{"subconscious":{"id":"subconscious","env":["SUBCONSCIOUS_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.subconscious.dev/v1","name":"Subconscious","doc":"https://docs.subconscious.dev","models":{"subconscious/glm-5.2":{"id":"subconscious/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"subconscious/tim-qwen3.6-27b":{"id":"subconscious/tim-qwen3.6-27b","name":"TIM-Qwen3.6 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":5000},"cost":{"input":0.3,"output":3,"cache_read":0.15}}}},"tokengo":{"id":"tokengo","env":["TOKENGO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokengo.com/v1","name":"TokenGo","doc":"https://www.tokengo.com/docs","models":{"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":2.65,"cache_read":0.2}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.098,"output":0.196,"cache_read":0.028}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.2174,"output":0.326,"cache_read":0.06}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.19,"output":0.71,"cache_read":0.06}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.025,"cache_read":0.015}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.89,"output":3.2647,"cache_read":0.2226}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"modelis":{"id":"modelis","env":["MODELIS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://modelishub.com/v1","name":"Modelis","doc":"https://modelishub.com/pricing","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.0983,"output":0.1966}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]},{"type":"toggle"}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","max"]},{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":3,"output":9}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.768,"output":3.072}}}},"bothub":{"id":"bothub","env":["BOTHUB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://openai.bothub.ru/v1","name":"Bothub","doc":"https://bothub.ru/models","models":{"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.61,"output":4.84}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1,"output":0.28}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.06,"output":0.37}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.44}},"nemotron-3-ultra-550b-a55b:free":{"id":"nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra (free)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemma-4-31b-it:free":{"id":"gemma-4-31b-it:free","name":"Gemma 4 31B IT (free)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.72,"output":5.41}}}},"greenpt":{"id":"greenpt","env":["GREENPT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.greenpt.ai/v1","name":"GreenPT","doc":"https://docs.greenpt.ai","models":{"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1596,"output":0.399,"cache_read":0.0456}},"glm-5.2-caveman-ultra":{"id":"glm-5.2-caveman-ultra","name":"GLM-5.2 Caveman Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-ponytail-ultra":{"id":"glm-5.2-ponytail-ultra","name":"GLM-5.2 Ponytail Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.2-honey-ultra":{"id":"glm-5.2-honey-ultra","name":"GLM-5.2 Honey Ultra","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The most aggressive tier, close to answer-only. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7524,"output":4.275,"cache_read":0.2508}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.1938,"output":1.129,"cache_read":0.0627}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.9006,"output":4.389,"cache_read":0.1881}},"green-l":{"id":"green-l","name":"Green L","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":0.912}},"devstral-2-123b-instruct-2512":{"id":"devstral-2-123b-instruct-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":16384},"cost":{"input":0.57,"output":2.736}},"glm-5.2-honey-lite":{"id":"glm-5.2-honey-lite","name":"GLM-5.2 Honey Lite","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":1.083}},"green-l-raw":{"id":"green-l-raw","name":"Green L Raw","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.285,"output":0.912}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.798,"output":4.959}},"glm-5.2-caveman":{"id":"glm-5.2-caveman","name":"GLM-5.2 Caveman","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3.762,"output":18.81,"cache_read":0.9405}},"gemma-3-27b-it":{"id":"gemma-3-27b-it","name":"Gemma 3 27B","description":"Google Gemma 3 multimodal model for chat, reasoning, and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40000,"output":8192},"cost":{"input":0.342,"output":0.684}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.342,"output":2.052}},"green-s":{"id":"green-s","name":"Green S","description":"GreenPT speech-to-text model for pre-recorded and live transcription","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":8192},"cost":{"input":0.00437,"output":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.127754,"output":0.511016,"cache_read":0.0255508}},"glm-5.2-caveman-lite":{"id":"glm-5.2-caveman-lite","name":"GLM-5.2 Caveman Lite","description":"glm-5.2 carrying a built-in ruleset that compresses prose, keeping code and technical detail verbatim. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"mistral-medium-3.5-128b":{"id":"mistral-medium-3.5-128b","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":2.052,"output":10.26}},"glm-5.2-ponytail":{"id":"glm-5.2-ponytail","name":"GLM-5.2 Ponytail","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"gemma4":{"id":"gemma4","name":"gemma4","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.57,"output":1.71}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.228,"output":0.456}},"glm-5.2-honey":{"id":"glm-5.2-honey","name":"GLM-5.2 Honey","description":"glm-5.2 carrying a built-in ruleset that compresses both generated code and prose. The middle tier, and the ruleset as its authors wrote it. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"status":"deprecated","cost":{"input":1.756,"output":5.518}},"glm-5.2-ponytail-lite":{"id":"glm-5.2-ponytail-lite","name":"GLM-5.2 Ponytail Lite","description":"glm-5.2 carrying a built-in ruleset that compresses generated code, preferring platform features over custom code. The gentlest tier: it cuts filler only and keeps the explanation intact. Same upstream model and price per token as glm-5.2, with fewer output tokens.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.254,"output":5.016,"cache_read":0.3135}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen3 235B MoE instruct model for long-context multilingual chat and reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":1.026,"output":3.078}},"green-r-raw":{"id":"green-r-raw","name":"Green R Raw","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.399,"output":1.083}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.228,"output":0.798}},"holo2-30b-a3b":{"id":"holo2-30b-a3b","name":"Holo2 30B A3B","description":"H Company Holo2 vision model for GUI navigation and computer-use agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11","last_updated":"2025-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":22016,"output":16384},"cost":{"input":0.399,"output":0.969}},"voxtral-small-24b-2507":{"id":"voxtral-small-24b-2507","name":"Voxtral Small 24B","description":"Mistral Voxtral audio-understanding model for speech and transcription tasks","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.228,"output":0.513}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.27754,"output":5.11016,"cache_read":0.319385}},"green-s-pro":{"id":"green-s-pro","name":"Green S Pro","description":"GreenPT advanced speech-to-text model with multilingual transcription support","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-02","last_updated":"2025-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":8192},"cost":{"input":0.00437,"output":0}},"kimi-k2.6-fast":{"id":"kimi-k2.6-fast","name":"Kimi K2.6 Fast","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":1.655,"output":8.778}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.285,"output":0.285}},"green-r":{"id":"green-r","name":"Green R","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.399,"output":1.083}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":16384},"cost":{"input":1.254,"output":1.254}}}},"qiniu-ai":{"id":"qiniu-ai","env":["QINIU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.qnaigc.com/v1","name":"Qiniu","doc":"https://developer.qiniu.com/aitokenapi","models":{"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":4096}},"kling-v2-6":{"id":"kling-v2-6","name":"Kling-V2 6","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-13","last_updated":"2026-01-13","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":99999999,"output":99999999}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-12","last_updated":"2025-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":4096}},"gemini-3.0-pro-image-preview":{"id":"gemini-3.0-pro-image-preview","name":"Gemini 3.0 Pro Image Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Gemini 2.5 Flash Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":32768,"output":8192}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-12","last_updated":"2025-09-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768}},"gemini-2.0-flash":{"id":"gemini-2.0-flash","name":"Gemini 2.0 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":8192}},"claude-3.5-sonnet":{"id":"claude-3.5-sonnet","name":"Claude 3.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-09","last_updated":"2025-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8200}},"doubao-seed-2.0-mini":{"id":"doubao-seed-2.0-mini","name":"Doubao Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen-Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":4096}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":64000}},"gemini-3.0-pro-preview":{"id":"gemini-3.0-pro-preview","name":"Gemini 3.0 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"doubao-seed-1.6":{"id":"doubao-seed-1.6","name":"Doubao-Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40000,"output":4096}},"mimo-v2-flash":{"id":"mimo-v2-flash","name":"Mimo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"gemini-2.0-flash-lite":{"id":"gemini-2.0-flash-lite","name":"Gemini 2.0 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":8192}},"claude-4.5-opus":{"id":"claude-4.5-opus","name":"Claude 4.5 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"Qwen 2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"claude-4.0-opus":{"id":"claude-4.0-opus","name":"Claude 4.0 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000}},"qwen3-max-preview":{"id":"qwen3-max-preview","name":"Qwen3 Max Preview","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-06","last_updated":"2025-09-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000}},"doubao-1.5-pro-32k":{"id":"doubao-1.5-pro-32k","name":"Doubao 1.5 Pro 32k","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":12000}},"doubao-seed-2.0-code":{"id":"doubao-seed-2.0-code","name":"Doubao Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-22","last_updated":"2026-02-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen2.5-vl-7b-instruct":{"id":"qwen2.5-vl-7b-instruct","name":"Qwen 2.5 VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"claude-4.1-opus":{"id":"claude-4.1-opus","name":"Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000}},"qwen3-30b-a3b-thinking-2507":{"id":"qwen3-30b-a3b-thinking-2507","name":"Qwen3 30b A3b Thinking 2507","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":126000,"output":32000}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen-vl-max-2025-01-25":{"id":"qwen-vl-max-2025-01-25","name":"Qwen VL-MAX-2025-01-25","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"doubao-seed-2.0-pro":{"id":"doubao-seed-2.0-pro","name":"Doubao Seed 2.0 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536}},"glm-4.5":{"id":"glm-4.5","name":"GLM 4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304}},"claude-3.5-haiku":{"id":"claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192}},"doubao-seed-1.6-thinking":{"id":"doubao-seed-1.6-thinking","name":"Doubao-Seed 1.6 Thinking","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"doubao-1.5-vision-pro":{"id":"doubao-1.5-vision-pro","name":"Doubao 1.5 Vision Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"doubao-seed-1.6-flash":{"id":"doubao-seed-1.6-flash","name":"Doubao-Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":80000}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen3 30b A3b Instruct 2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40000,"output":4096}},"qwen3-vl-30b-a3b-thinking":{"id":"qwen3-vl-30b-a3b-thinking","name":"Qwen3-Vl 30b A3b Thinking","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000}},"claude-3.7-sonnet":{"id":"claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen 3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235b A22B Instruct 2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":64000}},"deepseek-v3-0324":{"id":"deepseek-v3-0324","name":"DeepSeek-V3-0324","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"doubao-1.5-thinking-pro":{"id":"doubao-1.5-thinking-pro","name":"Doubao 1.5 Thinking Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000}},"gemini-3.0-flash-preview":{"id":"gemini-3.0-flash-preview","name":"Gemini 3.0 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"claude-4.5-sonnet":{"id":"claude-4.5-sonnet","name":"Claude 4.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"qwen-max-2025-01-25":{"id":"qwen-max-2025-01-25","name":"Qwen2.5-Max-2025-01-25","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-14","last_updated":"2025-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":4096}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"doubao-seed-2.0-lite":{"id":"doubao-seed-2.0-lite","name":"Doubao Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000}},"claude-4.5-haiku":{"id":"claude-4.5-haiku","name":"Claude 4.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":64000}},"claude-4.0-sonnet":{"id":"claude-4.0-sonnet","name":"Claude 4.0 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"deepseek-v3.1":{"id":"deepseek-v3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"stepfun-ai/gelab-zero-4b-preview":{"id":"stepfun-ai/gelab-zero-4b-preview","name":"Stepfun-Ai/Gelab Zero 4b Preview","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096}},"meituan/longcat-flash-lite":{"id":"meituan/longcat-flash-lite","name":"Meituan/Longcat-Flash-Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":320000}},"meituan/longcat-flash-chat":{"id":"meituan/longcat-flash-chat","name":"Meituan/Longcat-Flash-Chat","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-11-05","last_updated":"2025-11-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Stepfun/Step-3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"output":4096}},"xiaomi/mimo-v2-flash":{"id":"xiaomi/mimo-v2-flash","name":"Xiaomi/Mimo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax/Minimax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"Minimax/Minimax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"Minimax/Minimax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"Minimax/Minimax-M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":128000}},"deepseek/deepseek-math-v2":{"id":"deepseek/deepseek-math-v2","name":"Deepseek/Deepseek-Math-V2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":160000,"output":160000}},"deepseek/deepseek-v3.2-exp-thinking":{"id":"deepseek/deepseek-v3.2-exp-thinking","name":"DeepSeek/DeepSeek-V3.2-Exp-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.1-terminus-thinking":{"id":"deepseek/deepseek-v3.1-terminus-thinking","name":"DeepSeek/DeepSeek-V3.1-Terminus-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek/DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"deepseek/deepseek-v3.2-251201":{"id":"deepseek/deepseek-v3.2-251201","name":"Deepseek/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000}},"x-ai/grok-code-fast-1":{"id":"x-ai/grok-code-fast-1","name":"x-AI/Grok-Code-Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000}},"x-ai/grok-4-fast-reasoning":{"id":"x-ai/grok-4-fast-reasoning","name":"X-Ai/Grok-4-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast-non-reasoning":{"id":"x-ai/grok-4.1-fast-non-reasoning","name":"X-Ai/Grok 4.1 Fast Non Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast-reasoning":{"id":"x-ai/grok-4.1-fast-reasoning","name":"X-Ai/Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":20000000,"output":2000000}},"x-ai/grok-4-fast":{"id":"x-ai/grok-4-fast","name":"x-AI/Grok-4-Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-20","last_updated":"2025-09-20","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4-fast-non-reasoning":{"id":"x-ai/grok-4-fast-non-reasoning","name":"X-Ai/Grok-4-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"x-AI/Grok-4.1-Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"OpenAI/GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000}},"openai/gpt-5":{"id":"openai/gpt-5","name":"OpenAI/GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":100000}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-07","last_updated":"2025-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":100000}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"Z-Ai/GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"Z-AI/GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":200000}},"z-ai/autoglm-phone-9b":{"id":"z-ai/autoglm-phone-9b","name":"Z-Ai/Autoglm Phone 9b","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":12800,"output":4096}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"Z-Ai/GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}}}},"ambient":{"id":"ambient","env":["AMBIENT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ambient.xyz/v1","name":"Ambient","doc":"https://ambient.xyz","models":{"ambient/large":{"id":"ambient/large","name":"Ambient Large","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.6,"output":2,"cache_read":0.15,"cache_write":0}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.19,"output":1.14,"cache_read":0.03,"cache_write":0}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08,"cache_write":0}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.2,"output":4.2,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0,"cache_write":0}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.08,"output":0.18,"cache_read":0.016,"cache_write":0}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.028,"cache_write":0}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.2,"cache_write":0}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.69,"output":3.49,"cache_read":0.14,"cache_write":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.6,"output":2,"cache_read":0.15,"cache_write":0}}}},"agentrouter":{"id":"agentrouter","env":["AGENTROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://agentrouter.org/v1","name":"AgentRouter","doc":"https://agentrouter.org/docs/opencode.html","models":{"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://agentrouter.org/v1"}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://agentrouter.org/v1"}}}},"xiaomi-token-plan-cn":{"id":"xiaomi-token-plan-cn","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-cn.xiaomimimo.com/v1","name":"Xiaomi Token Plan (China)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"nano-gpt":{"id":"nano-gpt","env":["NANO_GPT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://nano-gpt.com/api/v1","name":"NanoGPT","doc":"https://docs.nano-gpt.com","models":{"glm-4.1v-thinking-flashx":{"id":"glm-4.1v-thinking-flashx","name":"GLM 4.1V Thinking FlashX","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"claude-opus-4-thinking:8192":{"id":"claude-opus-4-thinking:8192","name":"Claude 4 Opus Thinking (8K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat 2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"input":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"gemma-4-31b-it-garnet":{"id":"gemma-4-31b-it-garnet","name":"Garnet","description":"Garnet is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"exa-answer":{"id":"exa-answer","name":"Exa (Answer)","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2025-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"input":4096,"output":4096},"cost":{"input":2.5,"output":2.5}},"gemini-2.0-pro-exp-02-05":{"id":"gemini-2.0-pro-exp-02-05","name":"Gemini 2.0 Pro 0205","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2097152,"input":2097152,"output":8192},"cost":{"input":1.989,"output":7.956,"cache_read":0.49725}},"Meta-Llama-3-1-8B-Instruct-FP8":{"id":"Meta-Llama-3-1-8B-Instruct-FP8","name":"Llama 3.1 8B (decentralized)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.02,"output":0.03,"cache_read":0.01}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.17,"output":1.53,"cache_read":0.085}},"ernie-5.0-thinking-preview":{"id":"ernie-5.0-thinking-preview","name":"Ernie 5.0 Thinking Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1,"output":3.5,"cache_read":0.5}},"mercury-coder-small":{"id":"mercury-coder-small","name":"Mercury Coder Small","description":"Model by Inception AI. A diffusion large language model that runs incredibly quickly (500+ tokens/second) while matching Claude 3.5 Haiku and GPT-4o-mini. 1st in speed on Copilot arena, and matching 2nd in quality.","family":"mercury","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.25,"output":1,"cache_read":0.125}},"gemma-4-26b-a4b-it-luminous":{"id":"gemma-4-26b-a4b-it-luminous","name":"Luminous Mirror","description":"Luminous Mirror is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"claude-sonnet-4-thinking:32768":{"id":"claude-sonnet-4-thinking:32768","name":"Claude 4 Sonnet Thinking (32K)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"gemma-4-26b-a4b-it-shadowsiren":{"id":"gemma-4-26b-a4b-it-shadowsiren","name":"Shadow Siren","description":"Shadow Siren is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"auto-model-premium":{"id":"auto-model-premium","name":"Auto model (Premium)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"mistral-code-latest":{"id":"mistral-code-latest","name":"Mistral Code Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"qwen3.5-122b-a10b:thinking":{"id":"qwen3.5-122b-a10b:thinking","name":"Qwen3.5 122B A10B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.437,"output":3.496,"cache_read":0.103788}},"doubao-seed-1-6-250615":{"id":"doubao-seed-1-6-250615","name":"Doubao Seed 1.6","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.204,"output":0.51,"cache_read":0.102}},"Gemma-4-26B-A4B-MeroMero":{"id":"Gemma-4-26B-A4B-MeroMero","name":"Gemma 4 26B A4B MeroMero","description":"Gemma 4 26B A4B MeroMero is an NVFP4 multimodal mixture-of-experts fine-tune for emotive dialogue, relationship scenes, creative writing, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"qwen-max":{"id":"qwen-max","name":"Qwen 2.5 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":1.5997,"output":6.392,"cache_read":0.79985}},"claw-low":{"id":"claw-low","name":"Claw Low","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0.08333}},"doubao-seed-2-0-mini-260215":{"id":"doubao-seed-2-0-mini-260215","name":"Doubao Seed 2.0 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32000},"cost":{"input":0.0493,"output":0.4845,"cache_read":0.02465}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.27,"output":2.16,"cache_read":0.135}},"claude-opus-4-1-thinking:1024":{"id":"claude-opus-4-1-thinking:1024","name":"Claude 4.1 Opus Thinking (1K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"gemma-4-31b-it-gemsicle":{"id":"gemma-4-31b-it-gemsicle","name":"Gemsicle","description":"Gemsicle is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.7,"cache_read":0.04}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.225,"output":1.8,"cache_read":0.1125}},"claude-haiku-4-5-20251001-thinking":{"id":"claude-haiku-4-5-20251001-thinking","name":"Claude Haiku 4.5 Thinking","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"gemini-2.5-flash-preview-09-2025-thinking":{"id":"gemini-2.5-flash-preview-09-2025-thinking","name":"Gemini 2.5 Flash Preview (09/2025) – Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemma-4-26b-a4b-it-opusdistill":{"id":"gemma-4-26b-a4b-it-opusdistill","name":"Opus Distill","description":"Opus Distill is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"glm-4-plus-0111":{"id":"glm-4-plus-0111","name":"GLM 4 Plus 0111","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":9.996,"output":9.996,"cache_read":4.998}},"gemini-2.0-pro-reasoner":{"id":"gemini-2.0-pro-reasoner","name":"Gemini 2.0 Pro Reasoner","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-05","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":1.292,"output":4.998,"cache_read":0.323}},"Qwen3.5-27B-Queen-Derestricted":{"id":"Qwen3.5-27B-Queen-Derestricted","name":"Qwen3.5 27B Queen Derestricted","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"qwen-3.6-plus":{"id":"qwen-3.6-plus","name":"Qwen 3.6 Plus","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_read":0.0325,"cache_write":0.40625}},"Gemma-4-31B-Cognitive-Unshackled":{"id":"Gemma-4-31B-Cognitive-Unshackled","name":"Gemma 4 31B Cognitive Unshackled","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemini-2.5-pro-preview-06-05":{"id":"gemini-2.5-pro-preview-06-05","name":"Gemini 2.5 Pro Preview 0605","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"claude-sonnet-4-thinking:8192":{"id":"claude-sonnet-4-thinking:8192","name":"Claude 4 Sonnet Thinking (8K)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":8192},"cost":{"input":0.04998,"output":0.2006,"cache_read":0.02499}},"qwen3.7-flash:thinking":{"id":"qwen3.7-flash:thinking","name":"Qwen3.7 Flash Thinking","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen3.5-omni-plus":{"id":"qwen3.5-omni-plus","name":"Qwen3.5 Omni Plus","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"qwen3.5","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0,"output":0}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"asi1-mini":{"id":"asi1-mini","name":"ASI1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1,"output":1,"cache_read":0.5}},"gemini-2.5-pro-preview-03-25":{"id":"gemini-2.5-pro-preview-03-25","name":"Gemini 2.5 Pro Preview 0325","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"claude-sonnet-4-5-20250929-thinking":{"id":"claude-sonnet-4-5-20250929-thinking","name":"Claude Sonnet 4.5 Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"glm-4-air-0111":{"id":"glm-4-air-0111","name":"GLM 4 Air 0111","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-11","last_updated":"2025-01-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":0.1394,"output":0.1394,"cache_read":0.0697}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"qwen25-vl-72b-instruct":{"id":"qwen25-vl-72b-instruct","name":"Qwen25 VL 72b","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-10","last_updated":"2025-05-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":115200},"cost":{"input":0.69989,"output":0.69989,"cache_read":0.349945}},"qwen3.5-flash:thinking":{"id":"qwen3.5-flash:thinking","name":"Qwen3.5 Flash Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"command-a-reasoning-08-2025":{"id":"command-a-reasoning-08-2025","name":"Cohere Command A (08/2025)","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":8192},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"phi-4-multimodal-instruct":{"id":"phi-4-multimodal-instruct","name":"Phi 4 Multimodal","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.07,"output":0.11,"cache_read":0.035}},"claude-opus-4-thinking:32768":{"id":"claude-opus-4-thinking:32768","name":"Claude 4 Opus Thinking (32K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"qwen3.5-flash":{"id":"qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"mistral-code-agent-latest":{"id":"mistral-code-agent-latest","name":"Mistral Code Agent Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"Qwen3.5-27B-BlueStar-v3-Derestricted":{"id":"Qwen3.5-27B-BlueStar-v3-Derestricted","name":"Qwen3.5 27B BlueStar v3 Derestricted","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"deepseek-reasoner":{"id":"deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"input":64000,"output":65536},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"gemma-4-e4b-it":{"id":"gemma-4-e4b-it","name":"Gemma 4 E4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.04,"output":0.2,"cache_read":0.02}},"GLM-4.6-Derestricted-v5":{"id":"GLM-4.6-Derestricted-v5","name":"GLM 4.6 Derestricted v5","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.4,"output":1.5,"cache_read":0.2}},"doubao-seed-1-6-flash-250615":{"id":"doubao-seed-1-6-flash-250615","name":"Doubao Seed 1.6 Flash","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.0374,"output":0.374,"cache_read":0.0187}},"gemini-2.5-flash-lite-preview-06-17":{"id":"gemini-2.5-flash-lite-preview-06-17","name":"Gemini 2.5 Flash Lite Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"claude-opus-4-thinking":{"id":"claude-opus-4-thinking","name":"Claude 4 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"brave-research":{"id":"brave-research","name":"Brave (Research)","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-10","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":16384},"cost":{"input":5,"output":5}},"doubao-1.5-pro-256k":{"id":"doubao-1.5-pro-256k","name":"Doubao 1.5 Pro 256k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":16384},"cost":{"input":0.799,"output":1.445,"cache_read":0.3995}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"glm-z1-airx":{"id":"glm-z1-airx","name":"GLM Z1 AirX","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"ernie-5.1:thinking":{"id":"ernie-5.1:thinking","name":"ERNIE 5.1 Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2026-05-10","last_updated":"2026-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":119000,"input":119000,"output":64000},"cost":{"input":0.75,"output":3,"cache_read":0.75}},"qwen3.8-max:thinking":{"id":"qwen3.8-max:thinking","name":"Qwen3.8 Max Thinking","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"universal-summarizer":{"id":"universal-summarizer","name":"Universal Summarizer","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":30,"output":30}},"venice-uncensored":{"id":"venice-uncensored","name":"Venice Uncensored","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"venice","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-10-01","last_updated":"2025-02-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.4}},"gemini-2.5-flash-preview-09-2025":{"id":"gemini-2.5-flash-preview-09-2025","name":"Gemini 2.5 Flash Preview (09/2025)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"Claude 4 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"doubao-1.5-pro-32k":{"id":"doubao-1.5-pro-32k","name":"Doubao 1.5 Pro 32k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-11-20","last_updated":"2025-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.1343,"output":0.3349,"cache_read":0.06715}},"gemma-4-26b-a4b-it-moonlight":{"id":"gemma-4-26b-a4b-it-moonlight","name":"Moonlight Dusk","description":"Moonlight Dusk is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"deepseek-chat-cheaper":{"id":"deepseek-chat-cheaper","name":"DeepSeek V3/Chat Cheaper","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.1,"output":0.425,"cache_read":0.05}},"gemma-4-26b-a4b-it-darksoul":{"id":"gemma-4-26b-a4b-it-darksoul","name":"Dark Soul","description":"Dark Soul is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"Gemini 2.5 Flash Lite Preview (09/2025)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"Gemma-4-31B-Queen":{"id":"Gemma-4-31B-Queen","name":"Gemma 4 31B Queen","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"deepseek-r1-sambanova":{"id":"deepseek-r1-sambanova","name":"DeepSeek R1 Fast","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":4.998,"output":6.987,"cache_read":2.499}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"perplexity-academic-researcher":{"id":"perplexity-academic-researcher","name":"Perplexity Academic Researcher","description":"Sonar Reasoning Pro with Perplexity's academic search mode. Prioritizes scholarly and peer-reviewed sources from academic repositories and returns cited research synthesis.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":115200},"cost":{"input":2,"output":8,"cache_read":1}},"claude-sonnet-4-thinking:64000":{"id":"claude-sonnet-4-thinking:64000","name":"Claude 4 Sonnet Thinking (64K)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen3.7-max:thinking":{"id":"qwen3.7-max:thinking","name":"Qwen3.7 Max Thinking","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"gemma-4-26b-a4b-uncensored":{"id":"gemma-4-26b-a4b-uncensored","name":"Gemma 4 26B A4B Uncensored","description":"Gemma 4 26B A4B Uncensored is an FP8 open-weight multimodal mixture-of-experts model LoRA-tuned for fewer refusals across chat, coding, tool use, and long-context work.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"claude-opus-4-5-20251101:thinking":{"id":"claude-opus-4-5-20251101:thinking","name":"Claude 4.5 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"qwen3.5-27b:thinking":{"id":"qwen3.5-27b:thinking","name":"Qwen3.5 27B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.27,"output":2.16,"cache_read":0.135}},"qwen3.5-0.8b":{"id":"qwen3.5-0.8b","name":"Qwen3.5 0.8B","description":"Qwen3.5 0.8B is a lightweight open-weight multimodal model from Alibaba for fast reasoning, visual understanding, tool use, and JSON output.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-16","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.06,"output":0.12,"cache_read":0.03}},"doubao-seed-2-0-code-preview-260215":{"id":"doubao-seed-2-0-code-preview-260215","name":"Doubao Seed 2.0 Code Preview","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.782,"output":3.893,"cache_read":0.391}},"ernie-5.1":{"id":"ernie-5.1","name":"ERNIE 5.1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-05-10","last_updated":"2026-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":119000,"input":119000,"output":64000},"cost":{"input":0.75,"output":3,"cache_read":0.75}},"gemma-4-26b-a4b-it-chimerax":{"id":"gemma-4-26b-a4b-it-chimerax","name":"Chimera X","description":"Chimera X is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"sarvam-105b":{"id":"sarvam-105b","name":"Sarvam 105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":4096},"cost":{"input":0.054,"output":0.2124,"cache_read":0.0336}},"deepseek-chat":{"id":"deepseek-chat","name":"DeepSeek V3/Deepseek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.1,"output":0.425,"cache_read":0.05}},"holo3-35b-a3b":{"id":"holo3-35b-a3b","name":"Holo3-35B-A3B","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.25,"output":1.8,"cache_read":0.125}},"hermes-high":{"id":"hermes-high","name":"Hermes High","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"claw-high":{"id":"claw-high","name":"Claw High","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"holo3-35b-a3b:thinking":{"id":"holo3-35b-a3b:thinking","name":"Holo3-35B-A3B Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.25,"output":1.8,"cache_read":0.125}},"qwen3.5-omni-flash":{"id":"qwen3.5-omni-flash","name":"Qwen3.5 Omni Flash","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"qwen3.5","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":49152,"input":49152,"output":16384},"cost":{"input":0,"output":0}},"qwen3.5-35b-a3b:thinking":{"id":"qwen3.5-35b-a3b:thinking","name":"Qwen3.5 35B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.225,"output":1.8,"cache_read":0.1125}},"claude-sonnet-4-thinking:1024":{"id":"claude-sonnet-4-thinking:1024","name":"Claude 4 Sonnet Thinking (1K)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"doubao-1.5-vision-pro-32k":{"id":"doubao-1.5-vision-pro-32k","name":"Doubao 1.5 Vision Pro 32k","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-11-20","last_updated":"2025-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.459,"output":1.377,"cache_read":0.2295}},"doubao-seed-2-0-lite-260215":{"id":"doubao-seed-2-0-lite-260215","name":"Doubao Seed 2.0 Lite","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32000},"cost":{"input":0.1462,"output":0.8738,"cache_read":0.0731}},"Gemma-4-26B-A4B-MeroMero:thinking":{"id":"Gemma-4-26B-A4B-MeroMero:thinking","name":"Gemma 4 26B A4B MeroMero Thinking","description":"Gemma 4 26B A4B MeroMero with thinking enabled for more deliberate emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"gemini-2.5-flash-preview-04-17:thinking":{"id":"gemini-2.5-flash-preview-04-17:thinking","name":"Gemini 2.5 Flash Preview Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-04-17","last_updated":"2025-04-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":3.5,"cache_read":0.015}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"claw-medium":{"id":"claw-medium","name":"Claw Medium","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"glm-4-long":{"id":"glm-4-long","name":"GLM-4 Long","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":4096},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"Gemma-4-31B-GarnetV2":{"id":"Gemma-4-31B-GarnetV2","name":"Gemma 4 31B Garnet V2","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"gemini-2.5-flash-preview-04-17":{"id":"gemini-2.5-flash-preview-04-17","name":"Gemini 2.5 Flash Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-04-17","last_updated":"2025-04-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled":{"id":"Gemma-4-31B-Claude-4.6-Opus-Reasoning-Distilled","name":"Gemma 4 31B Claude 4.6 Opus Reasoning Distilled","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"claude","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.0306}},"qwen3.8-27b:thinking":{"id":"qwen3.8-27b:thinking","name":"Qwen3.8 27B Thinking","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.7,"cache_read":0.04}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":995904,"input":995904,"output":32768},"cost":{"input":0.3995,"output":1.2002,"cache_read":0.19975}},"fastgpt":{"id":"fastgpt","name":"Web Answer","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-23","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":7.5,"output":7.5}},"gemini-2.5-flash-nothinking":{"id":"gemini-2.5-flash-nothinking","name":"Gemini 2.5 Flash (No Thinking)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.437,"output":3.496,"cache_read":0.103788}},"deepseek-reasoner-cheaper":{"id":"deepseek-reasoner-cheaper","name":"Deepseek R1 Cheaper","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"sonar":{"id":"sonar","name":"Perplexity Simple","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"input":127072,"output":114364},"cost":{"input":1,"output":1,"cache_read":0.5}},"gemini-2.5-flash-lite-preview-09-2025-thinking":{"id":"gemini-2.5-flash-lite-preview-09-2025-thinking","name":"Gemini 2.5 Flash Lite Preview (09/2025) – Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"claude-sonnet-4-thinking":{"id":"claude-sonnet-4-thinking","name":"Claude 4 Sonnet Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"auto-model-standard":{"id":"auto-model-standard","name":"Auto model (Standard)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Gemini 3 Pro Image","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"input":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Perplexity Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":115200},"cost":{"input":2,"output":8,"cache_read":1}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax M1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-08","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.1394,"output":1.3328,"cache_read":0.0697}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.2,"output":0.5,"cache_read":0.1}},"Gemma-4-31B-DarkIdol":{"id":"Gemma-4-31B-DarkIdol","name":"Gemma 4 31B DarkIdol","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.306,"output":0.306,"cache_read":0.153}},"qwen3.7-plus:thinking":{"id":"qwen3.7-plus:thinking","name":"Qwen3.7 Plus Thinking","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"auto-model":{"id":"auto-model","name":"Auto model","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":0,"output":0}},"gemma-4-31b-it-darkidol":{"id":"gemma-4-31b-it-darkidol","name":"DarkIdol","description":"DarkIdol is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"claude-opus-4-1-thinking":{"id":"claude-opus-4-1-thinking","name":"Claude 4.1 Opus Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"glm-4.1v-thinking-flash":{"id":"glm-4.1v-thinking-flash","name":"GLM 4.1V Thinking Flash","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"qwen3.5-4b":{"id":"qwen3.5-4b","name":"Qwen3.5 4B","description":"Qwen3.5 4B is a compact open-weight multimodal model from Alibaba for reasoning, coding, visual understanding, tool use, and structured output.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-16","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"claude-opus-4-1-thinking:32768":{"id":"claude-opus-4-1-thinking:32768","name":"Claude 4.1 Opus Thinking (32K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"mistral-small-31-24b-instruct":{"id":"mistral-small-31-24b-instruct","name":"Mistral Small 31 24b Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":102400},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"gemma-4-31b-it-fabled":{"id":"gemma-4-31b-it-fabled","name":"Fabled","description":"Fabled is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"agnes-3.0-flash":{"id":"agnes-3.0-flash","name":"Agnes 3.0 Flash","description":"Agnes 3.0 Flash is a low-cost model for coding, tool use, and multi-turn agent tasks. It supports text and image input, optional thinking, and a 512K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":65536},"cost":{"input":0.05,"output":0.15,"cache_read":0.005}},"qwen3-vl-235b-a22b-instruct-original":{"id":"qwen3-vl-235b-a22b-instruct-original","name":"Qwen3 VL 235B A22B Instruct Original","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.25}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":245760,"input":245760,"output":65536},"cost":{"input":1.04,"output":6.24,"cache_read":0.52}},"gemini-2.5-flash-preview-05-20":{"id":"gemini-2.5-flash-preview-05-20","name":"Gemini 2.5 Flash 0520","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"input":1048000,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"gemma-4-31b-it-gembrain":{"id":"gemma-4-31b-it-gembrain","name":"Gembrain","description":"Gembrain is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"claude-opus-4-thinking:1024":{"id":"claude-opus-4-thinking:1024","name":"Claude 4 Opus Thinking (1K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"longcat-2.0:thinking":{"id":"longcat-2.0:thinking","name":"LongCat 2.0 Thinking","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"input":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"qwen3-vl-235b-a22b-thinking":{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.5,"output":6,"cache_read":0.25}},"celeris-1":{"id":"celeris-1","name":"Celeris 1","description":"Celeris 1 is a diffusion language model built for ultra-low-latency classification, extraction, judging, query rewriting, and other short structured responses.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-07-25","last_updated":"2026-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":2,"output":6,"cache_read":1}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"deepseek-v3-0324":{"id":"deepseek-v3-0324","name":"DeepSeek Chat 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.77,"cache_read":0.135}},"brave-pro":{"id":"brave-pro","name":"Brave (Pro)","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-10","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":5,"output":5}},"gemma-4-31b-it-novelist":{"id":"gemma-4-31b-it-novelist","name":"Novelist","description":"Novelist is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemma-4-12b-it":{"id":"gemma-4-12b-it","name":"Gemma 4 12B Instruct","description":"Google's Gemma 4 12B Instruct is an open-weight multimodal model for text, image, audio, and video understanding, with tool calling and structured output support.","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"sonar-pro":{"id":"sonar-pro","name":"Perplexity Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":3,"output":15,"cache_read":1.5}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"Qwen3.5 2B","description":"Qwen3.5 2B is a small open-weight multimodal model from Alibaba for efficient reasoning, coding, visual understanding, tool use, and JSON output.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-16","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.08,"output":0.16,"cache_read":0.04}},"doubao-seed-2-0-pro-260215":{"id":"doubao-seed-2-0-pro-260215","name":"Doubao Seed 2.0 Pro","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.782,"output":3.876,"cache_read":0.391}},"Gemma-4-31B-MeroMero-v2:thinking":{"id":"Gemma-4-31B-MeroMero-v2:thinking","name":"Gemma 4 31B MeroMero v2 Thinking","description":"Gemma 4 31B MeroMero v2 with thinking enabled for more deliberate emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":115200},"cost":{"input":3.4,"output":13.6,"cache_read":1.7}},"kimi-k2-instruct-fast":{"id":"kimi-k2-instruct-fast","name":"Kimi K2 0711 Fast","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-15","last_updated":"2025-07-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"gemma-4-12b-it-semancer":{"id":"gemma-4-12b-it-semancer","name":"Gemma 4 12B Semancer","description":"Gemma 4 12B Semancer is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 131,072-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"Gemma-4-31B-MeroMero-v2":{"id":"Gemma-4-31B-MeroMero-v2","name":"Gemma 4 31B MeroMero v2","description":"Gemma 4 31B MeroMero v2 is a LoRA finetune for emotive dialogue, relationship scenes, creative writing, and multimodal roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"auto-model-basic":{"id":"auto-model-basic","name":"Auto model (Basic)","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-04-16","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":1000000},"cost":{"input":9.996,"output":19.992,"cache_read":4.998}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":0.375}},"gemini-2.5-flash-preview-05-20:thinking":{"id":"gemini-2.5-flash-preview-05-20:thinking","name":"Gemini 2.5 Flash 0520 Thinking","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"input":1048000,"output":65536},"cost":{"input":0.15,"output":3.5,"cache_read":0.015}},"gemini-2.5-pro-exp-03-25":{"id":"gemini-2.5-pro-exp-03-25","name":"Gemini 2.5 Pro Experimental 0325","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"Claude 4 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"qwen-long":{"id":"qwen-long","name":"Qwen Long 10M","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-08-01","last_updated":"2025-01-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"input":10000000,"output":8192},"cost":{"input":0.1003,"output":0.408,"cache_read":0.05015}},"phi-4-mini-instruct":{"id":"phi-4-mini-instruct","name":"Phi 4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"hermes-low":{"id":"hermes-low","name":"Hermes Low","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0.08333}},"nano-gpt-help":{"id":"nano-gpt-help","name":"NanoGPT Help","description":"Text-only NanoGPT support assistant. Questions are processed by the Help inference provider; do not paste secrets or account credentials. Covers the website, models, API, pricing, memory, media generation, and support.","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-06","last_updated":"2026-06-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":6000,"input":6000,"output":512},"cost":{"input":0,"output":0}},"gemma-4-12b-it-station-keeper":{"id":"gemma-4-12b-it-station-keeper","name":"Gemma 4 12B StationKeeper","description":"Gemma 4 12B StationKeeper is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 131,072-token context window.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max 2026-01-23","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-01-26","last_updated":"2026-01-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.2002,"output":6.001,"cache_read":0.6001}},"gemini-2.5-pro-preview-05-06":{"id":"gemini-2.5-pro-preview-05-06","name":"Gemini 2.5 Pro Preview 0506","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2025-05-06","last_updated":"2025-05-06","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2.5,"output":10,"cache_read":0.25}},"brave":{"id":"brave","name":"Brave (Answers)","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-02-13","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":5,"output":5}},"gemma-4-26b-a4b-uncensored:thinking":{"id":"gemma-4-26b-a4b-uncensored:thinking","name":"Gemma 4 26B A4B Uncensored Thinking","description":"Gemma 4 26B A4B Uncensored with thinking enabled for more deliberate coding, multimodal analysis, tool use, and long-context problem solving.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"pokee-isaac":{"id":"pokee-isaac","name":"Pokee-Isaac 28B","description":"Pokee-Isaac is a 28B agentic model with a roughly 10-million-token context window, function calling, and OpenAI-compatible structured output. Pokee bills in $0.01 increments, rounding each non-zero request up to the next cent.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"input":10000000,"output":60000},"cost":{"input":0.15,"output":1,"cache_read":0.075}},"claude-opus-4-1-thinking:8192":{"id":"claude-opus-4-1-thinking:8192","name":"Claude 4.1 Opus Thinking (8K)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5}},"ernie-x1.1-preview":{"id":"ernie-x1.1-preview","name":"ERNIE X1.1","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"ernie","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":64000,"input":64000,"output":8192},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"gemma-4-e2b-it":{"id":"gemma-4-e2b-it","name":"Gemma 4 E2B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.02,"output":0.1,"cache_read":0.01}},"gemini-exp-1206":{"id":"gemini-exp-1206","name":"Gemini 2.0 Pro 1206","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2097152,"input":2097152,"output":8192},"cost":{"input":1.258,"output":4.998,"cache_read":0.629}},"gemma-4-31b-it-isometry":{"id":"gemma-4-31b-it-isometry","name":"Isometry","description":"Isometry is a multimodal Gemma 4 31B creative finetune for expressive dialogue, long-form storytelling, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"gemma-4-26b-a4b-it-musica":{"id":"gemma-4-26b-a4b-it-musica","name":"Musica","description":"Musica is a Gemma 4 26B A4B multimodal mixture-of-experts fine-tune for creative writing, expressive dialogue, and roleplay.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"hermes-medium":{"id":"hermes-medium","name":"Hermes Medium","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-11","last_updated":"2026-05-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude 4.5 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"deepclaude":{"id":"deepclaude","name":"DeepClaude","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-10","last_updated":"2025-02-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3}},"qvq-max":{"id":"qvq-max","name":"Qwen: QvQ Max","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-28","last_updated":"2025-03-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":1.2,"output":4.8,"cache_read":0.6}},"LatitudeGames/Wayfarer-Large-70B-Llama-3.3":{"id":"LatitudeGames/Wayfarer-Large-70B-Llama-3.3","name":"Llama 3.3 70B Wayfarer","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"qwen/qwen3.5-397b-a17b-thinking":{"id":"qwen/qwen3.5-397b-a17b-thinking","name":"Qwen3.5 397B A17B Thinking","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":258048,"input":258048,"output":65536},"cost":{"input":0.6,"output":3.6,"cache_read":0.3}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.5}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.15,"output":0.65,"cache_read":0.075}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.05,"output":0.15,"cache_read":0.025}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.15}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen 3 14b","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.08,"output":0.24,"cache_read":0.04}},"qwen/Qwen3-8B":{"id":"qwen/Qwen3-8B","name":"Qwen 3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.47,"output":0.47,"cache_read":0.235}},"qwen/qwen3.8-27b-obliterated":{"id":"qwen/qwen3.8-27b-obliterated","name":"Qwen 3.8 27B Obliterated","description":"Qwen 3.8 27B Obliterated is an open-weight multimodal model LoRA-tuned for fewer refusals across chat, coding, reasoning, tool use, and long-context work.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.8-27b-queen":{"id":"qwen/qwen3.8-27b-queen","name":"Qwen 3.8 27B Queen","description":"Qwen 3.8 27B Queen is an open-weight roleplay finetune with image understanding, tool calling, optional reasoning, and a 262,144-token context window.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.5-plus-thinking":{"id":"qwen/qwen3.5-plus-thinking","name":"Qwen3.5 Plus Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen 3 32b","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen 3 Coder 480B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"input":262000,"output":65536},"cost":{"input":0.13,"output":0.5,"cache_read":0.065}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.2,"output":1.5,"cache_read":0.1}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":258048,"input":258048,"output":65536},"cost":{"input":0.6,"output":3.6,"cache_read":0.3}},"qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen 3 235b A22B 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.13,"output":0.5,"cache_read":0.065}},"qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"qwen/qwen3.8-27b-fable":{"id":"qwen/qwen3.8-27b-fable","name":"Qwen 3.8 27B Fable","description":"Qwen 3.8 27B Fable is an open-weight multimodal creative finetune for expressive dialogue, long-form storytelling, character work, and roleplay.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-29","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.2002,"output":6.001,"cache_read":0.6001}},"qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3 Next 80B A3B (Instruct)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.15,"output":0.65,"cache_read":0.075}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B (Max)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":991000,"input":991000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.8-27b-uncensored":{"id":"qwen/qwen3.8-27b-uncensored","name":"Qwen 3.8 27B Uncensored","description":"Qwen 3.8 27B Uncensored is an NVFP4 open-weight multimodal model LoRA-tuned for fewer refusals across chat, coding, reasoning, tool use, and long-context work.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":41000,"input":41000,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen 3 235b A22B 2507 Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-07-03","last_updated":"2025-07-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.357,"output":0.408,"cache_read":0.1785}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen 3 235b A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"qwen/Qwen3.6-35B-A3B":{"id":"qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.112,"output":0.8,"cache_read":0.056}},"qwen/Qwen3.6-35B-A3B:thinking":{"id":"qwen/Qwen3.6-35B-A3B:thinking","name":"Qwen3.6 35B A3B Thinking","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.112,"output":0.8,"cache_read":0.056}},"qwen/qwen3.8-27b-uncensored:thinking":{"id":"qwen/qwen3.8-27b-uncensored:thinking","name":"Qwen 3.8 27B Uncensored Thinking","description":"Qwen 3.8 27B Uncensored with thinking enabled for more deliberate creative work, coding, multimodal analysis, tool use, and long-context problem solving.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":983616,"input":983616,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"qwen/qwen3.8-27b-obliterated:thinking":{"id":"qwen/qwen3.8-27b-obliterated:thinking","name":"Qwen 3.8 27B Obliterated Thinking","description":"Qwen 3.8 27B Obliterated with thinking enabled for more deliberate creative work, coding, multimodal analysis, tool use, and long-context problem solving.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.25,"output":1.5,"cache_read":0.125}},"qwen/Qwen2.5-Coder-32B-Instruct":{"id":"qwen/Qwen2.5-Coder-32B-Instruct","name":"Qwen 2.5 Coder 32b","description":"Open coding-focused Qwen model for code generation, repair, and repository reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"MarinaraSpaghetti/NemoMix-Unleashed-12B":{"id":"MarinaraSpaghetti/NemoMix-Unleashed-12B","name":"NemoMix 12B Unleashed","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Llama 3.1 8b (uncensored)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.8,"output":1.6,"cache_read":0.4}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion 3.0","description":"Aion 3.0 is a GLM-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion 3.0 Mini","description":"Aion 3.0 Mini is a DeepSeek-family collaborative generation model tuned for immersive roleplay and storytelling, with stronger narrative structure, tension, conflict, and nuanced mature themes.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"nothingiisreal/L3.1-70B-Celeste-V0.1-BF16":{"id":"nothingiisreal/L3.1-70B-Celeste-V0.1-BF16","name":"Llama 3.1 70B Celeste v0.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"ornith-ai/ornith-1.5-35b-a3b":{"id":"ornith-ai/ornith-1.5-35b-a3b","name":"Ornith 1.5 35B","description":"Ornith 1.5 35B A3B is an open-weight mixture-of-experts model for agentic coding, tool use, image understanding, and long-context work. This variant disables thinking for faster direct responses.","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"ornith-ai/ornith-1.5-35b-a3b:thinking":{"id":"ornith-ai/ornith-1.5-35b-a3b:thinking","name":"Ornith 1.5 35B Thinking","description":"Ornith 1.5 35B A3B is an open-weight mixture-of-experts model for agentic coding, reasoning, tool use, image understanding, and long-context work. This variant enables thinking by default.","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"abliteration-ai/abliterated-model-large":{"id":"abliteration-ai/abliterated-model-large","name":"Abliterated Model Large","description":"Abliteration.ai's large text reasoning model is derived from GLM-5.2 and supports native tool calling, structured output, automatic prompt caching, and a one-million-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliteration-ai/abliterated-model-large-v2":{"id":"abliteration-ai/abliterated-model-large-v2","name":"Abliterated Model Large V2","description":"Abliteration.ai's default large text reasoning model is derived from GLM-5.3 for harder reasoning and evaluation workloads, with automatic prompt caching and a one-million-token context window.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliteration-ai/abliterated-model":{"id":"abliteration-ai/abliterated-model","name":"Abliterated Model","description":"Abliteration.ai's multimodal reasoning model supports text and image input, structured output, automatic prompt caching, and a 262K-token context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":262134},"cost":{"input":3,"output":3,"cache_read":0.3}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":6144,"input":6144,"output":4096},"cost":{"input":0.799,"output":1.207,"cache_read":0.3995}},"chutesai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"chutesai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24b Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"chutesai","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.2,"output":0.4,"cache_read":0.1}},"stepfun-ai/step-3.5-flash":{"id":"stepfun-ai/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"stepfun-ai/step-3.5-flash-2603":{"id":"stepfun-ai/step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.05}},"pamanseau/OpenReasoning-Nemotron-32B":{"id":"pamanseau/OpenReasoning-Nemotron-32B","name":"OpenReasoning Nemotron 32B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"deepseek-ai/DeepSeek-V3.1:thinking":{"id":"deepseek-ai/DeepSeek-V3.1:thinking","name":"DeepSeek V3.1 Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.7,"cache_read":0.1}},"deepseek-ai/deepseek-v3.2-exp-thinking":{"id":"deepseek-ai/deepseek-v3.2-exp-thinking","name":"DeepSeek V3.2 Exp Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.7,"cache_read":0.1}},"deepseek-ai/DeepSeek-V3.1-Terminus:thinking":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus:thinking","name":"DeepSeek V3.1 Terminus (Thinking)","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.25,"output":0.7,"cache_read":0.125}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":32768},"cost":{"input":0.4,"output":1.7,"cache_read":0.2}},"deepseek-ai/deepseek-v3.2-exp":{"id":"deepseek-ai/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"input":163840,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.25,"output":0.7,"cache_read":0.125}},"VongolaChouko/Starcannon-Unleashed-12B-v1.0":{"id":"VongolaChouko/Starcannon-Unleashed-12B-v1.0","name":"Mistral Nemo Starcannon 12b v1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"poolside/laguna-s-2.1:thinking":{"id":"poolside/laguna-s-2.1:thinking","name":"Laguna S 2.1 Thinking","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"featherless-ai/Qwerky-72B":{"id":"featherless-ai/Qwerky-72B","name":"Qwerky 72B","description":"General-purpose chat model for instruction following, writing, and analysis","family":"qwerky","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8192},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"stepfun/step-3.7-flash:thinking":{"id":"stepfun/step-3.7-flash:thinking","name":"Step 3.7 Flash Thinking","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"mlabonne/NeuralDaredevil-8B-abliterated":{"id":"mlabonne/NeuralDaredevil-8B-abliterated","name":"Neural Daredevil 8B abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.44,"output":0.44,"cache_read":0.22}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large 2411","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":102400},"cost":{"input":2.006,"output":6.001,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"mistralai/mistral-large-3-675b-instruct-2512":{"id":"mistralai/mistral-large-3-675b-instruct-2512","name":"Mistral Large 3 675B","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":256000},"cost":{"input":1,"output":3,"cache_read":0.5}},"mistralai/Devstral-Small-2505":{"id":"mistralai/Devstral-Small-2505","name":"Mistral Devstral Small 2505","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.06,"output":0.06,"cache_read":0.03}},"mistralai/devstral-2-123b-instruct-2512":{"id":"mistralai/devstral-2-123b-instruct-2512","name":"Devstral 2 123B","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Mistral Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":26214},"cost":{"input":0.1989,"output":0.595,"cache_read":0.09945}},"mistralai/ministral-14b-instruct-2512":{"id":"mistralai/ministral-14b-instruct-2512","name":"Ministral 3 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05}},"mistralai/mistral-small-4-119b-2603:thinking":{"id":"mistralai/mistral-small-4-119b-2603:thinking","name":"Mistral Small 4 119B Thinking","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/mistral-small-4-119b-2603":{"id":"mistralai/mistral-small-4-119b-2603","name":"Mistral Small 4 119B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.4,"output":1.4,"cache_read":0.2}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.05}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"mistralai/mixtral-8x22b-instruct-v0.1":{"id":"mistralai/mixtral-8x22b-instruct-v0.1","name":"Mixtral 8x22B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Ministral 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-04","last_updated":"2025-12-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.075}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0.4,"output":2,"cache_read":0.2}},"ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0":{"id":"ReadyArt/MS3.2-The-Omega-Directive-24B-Unslop-v2.0","name":"Omega Directive 24B Unslop v2.0","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated":{"id":"huihui-ai/DeepSeek-R1-Distill-Llama-70B-abliterated","name":"DeepSeek R1 Llama 70B Abliterated","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated":{"id":"huihui-ai/DeepSeek-R1-Distill-Qwen-32B-abliterated","name":"DeepSeek R1 Qwen Abliterated","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":1.4,"output":1.4,"cache_read":0.7}},"huihui-ai/Llama-3.3-70B-Instruct-abliterated":{"id":"huihui-ai/Llama-3.3-70B-Instruct-abliterated","name":"Llama 3.3 70B Instruct abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"huihui-ai/Qwen2.5-32B-Instruct-abliterated":{"id":"huihui-ai/Qwen2.5-32B-Instruct-abliterated","name":"Qwen 2.5 32B Abliterated","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-06","last_updated":"2025-01-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"xiaomi/mimo-v2.5-pro-crof":{"id":"xiaomi/mimo-v2.5-pro-crof","name":"MiMo V2.5 Pro (Crof)","description":"MiMo V2.5 Pro is Xiaomi's long-context flagship general model for coding and agentic orchestration. This separately served variant is intended for users concerned about censorship on the regular Xiaomi MiMo V2.5 Pro, and it is included in the NanoGPT subscription.","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.4,"output":0.8,"cache_read":0.003}},"xiaomi/mimo-v2.5:thinking":{"id":"xiaomi/mimo-v2.5:thinking","name":"MiMo V2.5 Thinking","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"xiaomi/mimo-v2.5-pro-crof:thinking":{"id":"xiaomi/mimo-v2.5-pro-crof:thinking","name":"MiMo V2.5 Pro Thinking (Crof)","description":"MiMo V2.5 Pro with Xiaomi thinking enabled for coding, long-context reasoning, and agentic orchestration. This separately served thinking variant is intended for users concerned about censorship on the regular Xiaomi MiMo V2.5 Pro, and it is included in the NanoGPT subscription.","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.4,"output":0.8,"cache_read":0.003}},"xiaomi/mimo-v2.5-pro:thinking":{"id":"xiaomi/mimo-v2.5-pro:thinking","name":"MiMo V2.5 Pro Thinking","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0}},"shisa-ai/shisa-v2-llama3.3-70b":{"id":"shisa-ai/shisa-v2-llama3.3-70b","name":"Shisa V2 Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"shisa-ai/shisa-v2.1-llama3.3-70b":{"id":"shisa-ai/shisa-v2.1-llama3.3-70b","name":"Shisa V2.1 Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":4096},"cost":{"input":0.5,"output":0.5,"cache_read":0.25}},"minimax/minimax-m3:thinking":{"id":"minimax/minimax-m3:thinking","name":"MiniMax M3 Thinking","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.33,"output":1.32,"cache_read":0.165}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.315,"output":1.26,"cache_read":0.1575}},"minimax/minimax-m2.7-turbo":{"id":"minimax/minimax-m2.7-turbo","name":"MiniMax M2.7 Turbo","description":"Efficient MiniMax model for quick assistance, coding, and routine automation","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.3}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax M2-her","description":"MiniMax M2 variant tuned for conversational and character-driven agent interactions","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65532,"input":65532,"output":2048},"cost":{"input":0.302,"output":1.207,"cache_read":0.151}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax 01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000192,"input":1000192,"output":16384},"cost":{"input":0.1394,"output":1.122,"cache_read":0.0697}},"minimax/minimax-latest":{"id":"minimax/minimax-latest","name":"MiniMax Latest","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"input":512000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"Sao10K/L3-8B-Stheno-v3.2":{"id":"Sao10K/L3-8B-Stheno-v3.2","name":"Sao10K Stheno 8b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-11-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"Sao10K/L3.3-70B-Euryale-v2.3":{"id":"Sao10K/L3.3-70B-Euryale-v2.3","name":"Llama 3.3 70B Euryale","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":20480,"input":20480,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Sao10K/L3.1-70B-Euryale-v2.2":{"id":"Sao10K/L3.1-70B-Euryale-v2.2","name":"Llama 3.1 70B Euryale","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":20480,"input":20480,"output":16384},"cost":{"input":0.306,"output":0.357,"cache_read":0.153}},"Sao10K/L3.1-70B-Hanami-x1":{"id":"Sao10K/L3.1-70B-Hanami-x1","name":"Llama 3.1 70B Hanami","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"alibaba/qwen3.6-27b:thinking":{"id":"alibaba/qwen3.6-27b:thinking","name":"Qwen3.6 27B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":131072}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.203,"output":2.24,"cache_read":0.1015}},"alibaba/qwen3.6-27b":{"id":"alibaba/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":260096,"input":260096,"output":65536},"cost":{"input":0.203,"output":2.24,"cache_read":0.1015}},"alibaba/qwen3.8-max-0902":{"id":"alibaba/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.17,"cache_write":2.5}},"alibaba/qwen3.6-flash":{"id":"alibaba/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":65536},"cost":{"input":0.19,"output":1.16,"cache_read":0.02,"cache_write":0.24}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991808,"input":991808,"output":131072},"cost":{"input":0.14,"output":0.42,"cache_read":0.016,"cache_write":0.2}},"nvidia/nemotron-3-ultra-550b-a55b:thinking":{"id":"nvidia/nemotron-3-ultra-550b-a55b:thinking","name":"Nvidia Nemotron 3 Ultra 550B Thinking","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nvidia Nemotron 3.5 Lightning","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-3-super-120b-a12b:thinking":{"id":"nvidia/nemotron-3-super-120b-a12b:thinking","name":"Nvidia Nemotron 3 Super 120B Thinking","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"nvidia/nemotron-3.5-lightning:thinking":{"id":"nvidia/nemotron-3.5-lightning:thinking","name":"Nvidia Nemotron 3.5 Lightning Thinking","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nvidia Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":16384},"cost":{"input":0.05,"output":0.25,"cache_read":0.025}},"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF":{"id":"nvidia/Llama-3.1-Nemotron-70B-Instruct-HF","name":"Nvidia Nemotron 70b","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.357,"output":0.408,"cache_read":0.1785}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nvidia Nemotron 3 Ultra 550B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"nvidia/Llama-3.3-Nemotron-Super-49B-v1":{"id":"nvidia/Llama-3.3-Nemotron-Super-49B-v1","name":"Nvidia Nemotron Super 49B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.15,"output":0.15,"cache_read":0.075}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nvidia Nemotron 3 Nano 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.6:thinking":{"id":"anthropic/claude-opus-4.6:thinking","name":"Claude 4.6 Opus Thinking","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-fable-latest":{"id":"anthropic/claude-fable-latest","name":"Claude Fable Latest","description":"Compatibility alias for Claude Fable.","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude 4.7 Opus","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6:thinking:low":{"id":"anthropic/claude-opus-4.6:thinking:low","name":"Claude 4.6 Opus Thinking Low","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-latest":{"id":"anthropic/claude-opus-latest","name":"Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.6:thinking:medium":{"id":"anthropic/claude-opus-4.6:thinking:medium","name":"Claude 4.6 Opus Thinking Medium","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-5:thinking":{"id":"anthropic/claude-sonnet-5:thinking","name":"Claude Sonnet 5 Thinking","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude 4.6 Opus","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-latest":{"id":"anthropic/claude-haiku-latest","name":"Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-sonnet-4.6:thinking":{"id":"anthropic/claude-sonnet-4.6:thinking","name":"Claude Sonnet 4.6 Thinking","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-sonnet-latest":{"id":"anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4.8:thinking":{"id":"anthropic/claude-opus-4.8:thinking","name":"Claude Opus 4.8 Thinking","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.6:thinking:max":{"id":"anthropic/claude-opus-4.6:thinking:max","name":"Claude 4.6 Opus Thinking Max","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.7:thinking":{"id":"anthropic/claude-opus-4.7:thinking","name":"Claude 4.7 Opus Thinking","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.12,"output":0.38,"cache_read":0.06}},"google/gemini-pro-latest":{"id":"google/gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.5-flash-thinking":{"id":"google/gemini-3.5-flash-thinking","name":"Gemini 3.5 Flash Thinking","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemini-3-flash-preview-thinking":{"id":"google/gemini-3-flash-preview-thinking","name":"Gemini 3 Flash Thinking","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro (Preview Custom Tools)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemini-3.1-pro-preview-high":{"id":"google/gemini-3.1-pro-preview-high","name":"Gemini 3.1 Pro (Preview High)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-21","last_updated":"2026-02-21","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google/gemma-4-31b-it:thinking":{"id":"google/gemma-4-31b-it:thinking","name":"Gemma 4 31B Thinking","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.1,"output":0.35,"cache_read":0.05}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"google/gemma-4-26b-a4b-it:thinking":{"id":"google/gemma-4-26b-a4b-it:thinking","name":"Gemma 4 26B A4B Thinking","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.13,"output":0.4,"cache_read":0.065}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.075}},"google/gemini-3.1-pro-preview-low":{"id":"google/gemini-3.1-pro-preview-low","name":"Gemini 3.1 Pro (Preview Low)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-21","last_updated":"2026-02-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"baseten/Kimi-K2-Instruct-FP4":{"id":"baseten/Kimi-K2-Instruct-FP4","name":"Kimi K2 0711 Instruct FP4","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"crofai/greg-2-super":{"id":"crofai/greg-2-super","name":"Greg 2 Super","description":"Greg 2 Super is CrofAI's balanced Greg 2 model for strong UI design, frontend iteration, coding, writing, and everyday agent tasks at a lower cost than Ultra.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-19","last_updated":"2026-06-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"input":229376,"output":229376},"cost":{"input":1.5,"output":5,"cache_read":0.25}},"crofai/greg-2-ultra":{"id":"crofai/greg-2-ultra","name":"Greg 2 Ultra","description":"Greg 2 Ultra is CrofAI's most capable Greg 2 model, tuned for premium UI design, agentic coding, creative writing, and higher-end general reasoning tasks.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-06-19","last_updated":"2026-06-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"input":229376,"output":229376},"cost":{"input":3,"output":10,"cache_read":0.5}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling:thinking":{"id":"thinkingmachines/inkling:thinking","name":"Inkling Thinking","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"input":1048000,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"input":1048000,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"thinkingmachines/Inkling-Small:thinking":{"id":"thinkingmachines/Inkling-Small:thinking","name":"Inkling Small Thinking","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"input":524288,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor (Data Used for Training)","description":"A much cheaper opt-in version of Muse Spark 1.2 with the same multimodal coding and agentic capabilities. Prompts and outputs sent to this Contributor model may be used by Meta for training and to improve its products; use the standard Muse Spark 1.2 model if you do not want your data used for training.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Meta's Muse Spark 1.3 Contributor is a frontier multimodal reasoning model for long-horizon coding and agentic workflows, with strong gains in computer use, browsing, professional tool use, codebase understanding, and million-token retrieval. It accepts text, images, audio, video, and files, supports tool calling and structured output, and always reasons before answering. Prompts and outputs may be used by Meta for training and to improve its products.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65536},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.15,"output":1.5,"cache_read":0.075}},"Steelskull/L3.3-Cu-Mai-R1-70b":{"id":"Steelskull/L3.3-Cu-Mai-R1-70b","name":"Llama 3.3 70B Cu Mai","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-Electra-R1-70b":{"id":"Steelskull/L3.3-Electra-R1-70b","name":"Steelskull Electra R1 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.69989,"output":0.69989,"cache_read":0.349945}},"Steelskull/L3.3-Nevoria-R1-70b":{"id":"Steelskull/L3.3-Nevoria-R1-70b","name":"Steelskull Nevoria R1 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Steelskull/L3.3-MS-Nevoria-70b":{"id":"Steelskull/L3.3-MS-Nevoria-70b","name":"Steelskull Nevoria 70b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"bytedance/doubao-seed-2.1-turbo":{"id":"bytedance/doubao-seed-2.1-turbo","name":"Doubao Seed 2.1 Turbo","description":"Fast, lower-cost model in the Doubao Seed 2.1 family for everyday chat, coding assistance, document work, and high-throughput productivity tasks. Supports a 256k context window and up to 128k output tokens. Note: privacy and logging guarantees may be limited.","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"bytedance/doubao-seed-2.1-pro":{"id":"bytedance/doubao-seed-2.1-pro","name":"Doubao Seed 2.1 Pro","description":"Higher-capability model in the Doubao Seed 2.1 family for agentic coding, long-context analysis, complex instruction following, and productivity workflows. Supports a 256k context window and up to 128k output tokens. Note: privacy and logging guarantees may be limited.","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":128000},"cost":{"input":1,"output":5,"cache_read":0.5}},"bytedance/doubao-seed-character":{"id":"bytedance/doubao-seed-character","name":"Doubao Seed Character","description":"ByteDance's character-focused Doubao Seed model for roleplay, persona consistency, dialogue, and creative character interactions. It supports text and image input with a 128k context window. Requests route through ZenMux to ByteDance; ZenMux does not publish a model-API zero-retention or training guarantee, so avoid sensitive data.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"release_date":"2026-07-18","last_updated":"2026-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.1179,"output":0.2947,"cache_read":0.0236,"cache_write":0.0025}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"ByteDance Seed 2.1 Turbo","description":"ByteDance Seed 2.1 Turbo is a multimodal model for coding and long-horizon agent workflows, including end-to-end software delivery and multi-step task execution. It supports text, image, and video input with a 262k context window.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":235929},"cost":{"input":0.5,"output":2.5,"cache_read":0.25}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"ByteDance Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.25}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"ByteDance Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.25,"output":2,"cache_read":0.125}},"NeverSleep/Lumimaid-v0.2-70B":{"id":"NeverSleep/Lumimaid-v0.2-70B","name":"Lumimaid v0.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":1,"output":1.5,"cache_read":0.5}},"TEE/qwen3.5-27b":{"id":"TEE/qwen3.5-27b","name":"Qwen3.5 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.3,"output":2.4,"cache_read":0.15}},"TEE/qwen3.8-27b":{"id":"TEE/qwen3.8-27b","name":"Qwen3.8 27B TEE","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.4,"output":3,"cache_read":0.15}},"TEE/kimi-k2.6":{"id":"TEE/kimi-k2.6","name":"Kimi K2.6 TEE","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":1.5,"output":5.25,"cache_read":0.375}},"TEE/glm-5.2":{"id":"TEE/glm-5.2","name":"GLM 5.2 TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.7}},"TEE/gemma4-31b":{"id":"TEE/gemma4-31b","name":"Gemma 4 31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-04","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.4,"output":1,"cache_read":0.4}},"TEE/deepseek-v4-flash":{"id":"TEE/deepseek-v4-flash","name":"DeepSeek V4 Flash TEE","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":393216},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"TEE/kimi-k2.7-code":{"id":"TEE/kimi-k2.7-code","name":"Kimi K2.7 Code TEE","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"TEE/gpt-oss-20b":{"id":"TEE/gpt-oss-20b","name":"GPT-OSS 20B TEE","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.04,"output":0.15,"cache_read":0.02}},"TEE/qwen2.5-vl-72b-instruct":{"id":"TEE/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"TEE/gemma4-31b:thinking":{"id":"TEE/gemma4-31b:thinking","name":"Gemma 4 31B Thinking TEE","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-02","last_updated":"2026-05-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":131072},"cost":{"input":0.4,"output":1,"cache_read":0.4}},"TEE/qwen3.5-397b-a17b":{"id":"TEE/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B TEE","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.55,"output":3.5,"cache_read":0.275}},"TEE/gemma-4-26b-a4b-uncensored":{"id":"TEE/gemma-4-26b-a4b-uncensored","name":"Gemma 4 26B A4B Uncensored TEE","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-23","last_updated":"2026-05-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":65536},"cost":{"input":0.15,"output":0.7,"cache_read":0.075}},"TEE/qwen3.6-27b":{"id":"TEE/qwen3.6-27b","name":"Qwen3.6 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.32,"output":2.7,"cache_read":0.16}},"TEE/kimi-k3":{"id":"TEE/kimi-k3","name":"Kimi K3 TEE","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":65535},"cost":{"input":3,"output":15,"cache_read":1.5}},"TEE/deepseek-v3.2":{"id":"TEE/deepseek-v3.2","name":"DeepSeek V3.2 TEE","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"input":164000,"output":65536},"cost":{"input":0.5,"output":1,"cache_read":0.25}},"TEE/qwen3.6-35b-a3b":{"id":"TEE/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B TEE","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.2,"output":1.27,"cache_read":0.1}},"TEE/glm-5.3-flash":{"id":"TEE/glm-5.3-flash","name":"GLM 5.3 Flash TEE","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"TEE/muse-glimmer-30b":{"id":"TEE/muse-glimmer-30b","name":"Muse Glimmer 30B TEE","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"TEE/gemma-4-31b-it":{"id":"TEE/gemma-4-31b-it","name":"Gemma 4 31B IT TEE","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.15,"output":0.46,"cache_read":0.075}},"TEE/qwen3.6-35b-a3b-uncensored":{"id":"TEE/qwen3.6-35b-a3b-uncensored","name":"Qwen3.6 35B A3B Uncensored TEE","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":131072}],"tool_call":true,"structured_output":true,"release_date":"2026-05-23","last_updated":"2026-05-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":131072},"cost":{"input":0.3,"output":1.5,"cache_read":0.15}},"TEE/qwen3-8b":{"id":"TEE/qwen3-8b","name":"Qwen3 8B TEE","description":"Qwen3 8B is an open-weight dense language model from Alibaba's Qwen team for efficient dialogue, reasoning, mathematics, coding, and tool use. Running inside a TEE (Trusted Execution Environment), with provider attestation support.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"input":40960,"output":8192},"cost":{"input":0.11,"output":0.45,"cache_read":0.11}},"TEE/glm-5.1":{"id":"TEE/glm-5.1","name":"GLM 5.1 TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":202752,"output":65535},"cost":{"input":1.5,"output":5.25,"cache_read":0.3}},"TEE/glm-5.2:thinking":{"id":"TEE/glm-5.2:thinking","name":"GLM 5.2 Thinking TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.7}},"TEE/gpt-oss-120b":{"id":"TEE/gpt-oss-120b","name":"GPT-OSS 120B TEE","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":2,"output":2,"cache_read":2}},"TEE/llama3-3-70b":{"id":"TEE/llama3-3-70b","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-03","last_updated":"2025-07-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":1.75,"output":2.75,"cache_read":1.75}},"TEE/glm-5.1-thinking":{"id":"TEE/glm-5.1-thinking","name":"GLM 5.1 Thinking TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":202752,"output":65535},"cost":{"input":1.5,"output":5.25,"cache_read":0.3}},"TEE/glm-5.3":{"id":"TEE/glm-5.3","name":"GLM 5.3 TEE","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"meganova-ai/manta-mini-1.0":{"id":"meganova-ai/manta-mini-1.0","name":"Manta Mini 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.02,"output":0.16,"cache_read":0.01}},"meganova-ai/manta-flash-1.0":{"id":"meganova-ai/manta-flash-1.0","name":"Manta Flash 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":16384},"cost":{"input":0.02,"output":0.16,"cache_read":0.01}},"meganova-ai/manta-pro-1.0":{"id":"meganova-ai/manta-pro-1.0","name":"Manta Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-20","last_updated":"2025-12-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"input":65536,"output":32768},"cost":{"input":0.06,"output":0.5,"cache_read":0.03}},"inception/mercury-2.5-preview":{"id":"inception/mercury-2.5-preview","name":"Mercury 2.5 Preview","description":"Mercury 2.5 Preview is Inception's latest and most intelligent diffusion language model. Instead of generating tokens strictly one at a time, it produces and refines multiple tokens in parallel, reaching up to 1,107 tokens per second on standard GPUs. It delivers a 10+ point intelligence gain over Mercury 2, with tunable reasoning, parallel tool calls, schema-aligned JSON output, and a 260K context window. It is built for latency-sensitive production work such as search agents, voice pipelines, customer support, rapid coding iteration, and coding subagents.","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"input":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"unsloth/gemma-3-4b-it":{"id":"unsloth/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2006,"output":0.2006,"cache_read":0.1003}},"unsloth/gemma-3-27b-it":{"id":"unsloth/gemma-3-27b-it","name":"Gemma 3 27B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":96000},"cost":{"input":0.2992,"output":0.2992,"cache_read":0.1496}},"unsloth/gemma-3-12b-it":{"id":"unsloth/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"unsloth","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.272,"output":0.272,"cache_read":0.136}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"inflatebot/MN-12B-Mag-Mell-R1":{"id":"inflatebot/MN-12B-Mag-Mell-R1","name":"Mag Mell R1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"deepcogito/cogito-v1-preview-qwen-32B":{"id":"deepcogito/cogito-v1-preview-qwen-32B","name":"Cogito v1 Preview Qwen 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-10","last_updated":"2025-05-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":1.8,"output":1.8,"cache_read":0.9}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":16384},"cost":{"input":5.25,"output":31.5,"cache_read":0.525}},"sakana/fugu-ultra-v1.1":{"id":"sakana/fugu-ultra-v1.1","name":"Fugu Ultra v1.1","description":"Sakana AI's upgraded Fugu Ultra release with stronger coding, agentic task execution, and advanced reasoning through dynamic orchestration of frontier models.","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":16384},"cost":{"input":5.25,"output":31.5,"cache_read":0.525}},"Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B":{"id":"Envoid/Llama-3.05-Nemotron-Tenyxchat-Storybreaker-70B","name":"Nemotron Tenyxchat Storybreaker 70b","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B":{"id":"Envoid/Llama-3.05-NT-Storybreaker-Ministral-70B","name":"Llama 3.05 Storybreaker Ministral 70b","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"NousResearch/hermes-3-llama-3.1-70b":{"id":"NousResearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-01-07","last_updated":"2026-01-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":8192},"cost":{"input":0.408,"output":0.408,"cache_read":0.204}},"NousResearch/hermes-4-405b":{"id":"NousResearch/hermes-4-405b","name":"Hermes 4 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"NousResearch/hermes-4-405b:thinking":{"id":"NousResearch/hermes-4-405b:thinking","name":"Hermes 4 Large (Thinking)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.15}},"failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5":{"id":"failspy/Meta-Llama-3-70B-Instruct-abliterated-v3.5","name":"Llama 3 70B abliterated","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":8192},"cost":{"input":0.7,"output":0.7,"cache_read":0.35}},"GalrionSoftworks/MN-LooseCannon-12B-v1":{"id":"GalrionSoftworks/MN-LooseCannon-12B-v1","name":"MN-LooseCannon-12B-v1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"IBM Granite 4.2 8B is an Apache 2.0-licensed dense model with native step-by-step reasoning and specialized training for agentic work. It can plan before acting, sequence tools, navigate codebases, work in terminals, and verify results across coding, search, mathematics, science, and complex instruction-following tasks.","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.1,"output":0.15,"cache_read":0.05}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-flash-latest":{"id":"deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Compatibility alias that routes to the newest dated DeepSeek V4 Flash release. Currently routes to DeepSeek V4 Flash 0731. ⚠️ This route goes directly to DeepSeek, so privacy and logging guarantees are limited.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-v4-flash-0731:thinking":{"id":"deepseek/deepseek-v4-flash-0731:thinking","name":"DeepSeek V4 Flash 0731 (Thinking)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"deepseek/deepseek-v4-flash:thinking":{"id":"deepseek/deepseek-v4-flash:thinking","name":"DeepSeek V4 Flash (Thinking)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro-0813:thinking":{"id":"deepseek/deepseek-v4-pro-0813:thinking","name":"DeepSeek V4 Pro 0813 Thinking","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4.1-flash:thinking":{"id":"deepseek/deepseek-v4.1-flash:thinking","name":"DeepSeek V4.1 Flash Thinking","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro:thinking":{"id":"deepseek/deepseek-v4-pro:thinking","name":"DeepSeek V4 Pro (Thinking)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163000,"input":163000,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"deepseek/deepseek-latest":{"id":"deepseek/deepseek-latest","name":"DeepSeek Latest","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.5,"cache_read":0.04}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":384000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"deepseek/deepseek-v3.2:thinking":{"id":"deepseek/deepseek-v3.2:thinking","name":"DeepSeek V3.2 Thinking","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163000,"input":163000,"output":65536},"cost":{"input":0.28,"output":0.42,"cache_read":0.14}},"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0":{"id":"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.0","name":"EVA Llama 3.33 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":2.006,"output":2.006,"cache_read":1.003}},"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1":{"id":"EVA-UNIT-01/EVA-LLaMA-3.33-70B-v0.1","name":"EVA-LLaMA-3.33-70B-v0.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":16384},"cost":{"input":2.006,"output":2.006,"cache_read":1.003}},"EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2":{"id":"EVA-UNIT-01/EVA-Qwen2.5-32B-v0.2","name":"EVA-Qwen2.5-32B-v0.2","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.799,"output":0.799,"cache_read":0.3995}},"EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2":{"id":"EVA-UNIT-01/EVA-Qwen2.5-72B-v0.2","name":"EVA-Qwen2.5-72B-v0.2","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.799,"output":0.799,"cache_read":0.3995}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Amazon Nova 2 Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":65535},"cost":{"input":0.51,"output":4.25,"cache_read":0.255}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Amazon Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"input":300000,"output":32000},"cost":{"input":0.799,"output":3.196,"cache_read":0.3995}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Amazon Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"input":300000,"output":5120},"cost":{"input":0.0595,"output":0.238,"cache_read":0.02975}},"LLM360/K2-Think":{"id":"LLM360/K2-Think","name":"K2-Think","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.17,"output":0.68,"cache_read":0.085}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Ling-3.0-flash is a 124B-parameter Mixture-of-Experts model with approximately 5.1B parameters active per token. It prioritizes token efficiency and production-scale agentic inference, helping coding and tool-using agents complete more work within constrained latency and serving budgets.","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash:thinking":{"id":"inclusionai/ling-3.0-flash:thinking","name":"Ling 3.0 Flash Thinking","description":"Ling-3.0-flash Thinking enables visible reasoning on inclusionAI's token-efficient 124B-parameter Mixture-of-Experts model for harder coding, tool use, planning, and production-scale agent workflows.","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.075,"output":0.22,"cache_read":0.015}},"inclusionai/ling-3.0-flash-vl":{"id":"inclusionai/ling-3.0-flash-vl","name":"Ling 3.0 Flash VL","description":"Ling 3.0 Flash VL is inclusionAI's native multimodal Mixture-of-Experts model with 124B total parameters and 5.5B active parameters per token. It combines image and video understanding with reasoning and tool use for document analysis, charts, visual verification, and interface-based agent tasks. Thinking is enabled by default and can be turned off in settings.","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"release_date":"2026-09-09","last_updated":"2026-09-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"MiniMaxAI/MiniMax-M1-80k":{"id":"MiniMaxAI/MiniMax-M1-80k","name":"MiniMax M1 80K","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-01-08","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":131072},"cost":{"input":0.6052,"output":2.4225,"cache_read":0.3026}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":2.006,"output":2.992,"cache_read":1.003}},"anthracite-org/magnum-v2-72b":{"id":"anthracite-org/magnum-v2-72b","name":"Magnum V2 72B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":2.006,"output":2.992,"cache_read":1.003}},"Salesforce/Llama-xLAM-2-70b-fc-r":{"id":"Salesforce/Llama-xLAM-2-70b-fc-r","name":"Llama-xLAM-2 70B fc-r","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-13","last_updated":"2025-04-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":2.5,"cache_read":1.25}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"input":2000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":1}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"input":2000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":1}},"x-ai/grok-latest":{"id":"x-ai/grok-latest","name":"Grok Latest","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama 3.1 8b Instruct","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.0544,"output":0.085,"cache_read":0.0272}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3b Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-09-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":8192},"cost":{"input":0.0306,"output":0.0493,"cache_read":0.0153}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":328000,"input":328000,"output":65536},"cost":{"input":0.085,"output":0.46,"cache_read":0.0425}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama 3.3 70b Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":16384},"cost":{"input":0.05,"output":0.23,"cache_read":0.025}},"abacusai/Dracarys-72B-Instruct":{"id":"abacusai/Dracarys-72B-Instruct","name":"Llama 3.1 70B Dracarys 2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-02","last_updated":"2025-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"Gryphe/MythoMax-L2-13b":{"id":"Gryphe/MythoMax-L2-13b","name":"MythoMax 13B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"input":4096,"output":3686},"cost":{"input":0.1003,"output":0.1003,"cache_read":0.05015}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"OpenAI o4-mini high","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-12-04","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT 5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/o3-mini-low":{"id":"openai/o3-mini-low","name":"OpenAI o3-mini (Low)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-01-31","last_updated":"2025-01-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3-pro-2025-06-10":{"id":"openai/o3-pro-2025-06-10","name":"OpenAI o3-pro (2025-06-10)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":22,"output":88,"cache_read":11}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT 4.1 Nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT 5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":15,"output":120,"cache_read":1.5}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"OpenAI o3-mini (High)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-01-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT 5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"GPT 6 Astra Pro","description":"GPT 6 Astra in Pro reasoning mode. Uses additional model work for difficult tasks, with higher latency and token usage at the same per-token rates. Reasoning effort remains independently configurable.","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT 5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT 5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.1-2025-11-13":{"id":"openai/gpt-5.1-2025-11-13","name":"GPT-5.1 (2025-11-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-latest":{"id":"openai/gpt-latest","name":"GPT Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-03-29","last_updated":"2026-03-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT 6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT 5.6 Luna Pro","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT 4.1 Mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT 5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.2,"output":0.3}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT 5.6 Sol Pro","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"release_date":"2026-02-23","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.075,"output":0.3}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT 5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT 5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/o1":{"id":"openai/o1","name":"OpenAI o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT 5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"OpenAI o1 Pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":150,"output":600,"cache_read":75}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT 4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT 5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT 5.6 Terra Pro","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT 5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"input":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT 5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0.35,"output":0.75}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT 5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT 5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"OpenAI o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3-mini":{"id":"openai/o3-mini","name":"OpenAI o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"OpenAI o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":1}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT 5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":1050000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"moonshotai/kimi-latest":{"id":"moonshotai/kimi-latest","name":"Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":2,"output":10,"cache_read":0.2}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code High-Speed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":1.9,"output":8,"cache_read":0.32}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.5,"output":2.6,"cache_read":0.125}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":100352},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":98304},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":943718},"cost":{"input":2,"output":10,"cache_read":0.2}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"moonshotai/kimi-k2.6:thinking":{"id":"moonshotai/kimi-k2.6:thinking","name":"Kimi K2.6 Thinking","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.5,"output":2.6,"cache_read":0.125}},"moonshotai/kimi-k2-instruct-0711":{"id":"moonshotai/kimi-k2-instruct-0711","name":"Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-01-01","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.4,"output":1.8,"cache_read":0.2}},"moonshotai/kimi-k2.5:thinking":{"id":"moonshotai/kimi-k2.5:thinking","name":"Kimi K2.5 Thinking","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65536},"cost":{"input":0.3,"output":1.9,"cache_read":0.15}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Cohere: Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":4096},"cost":{"input":2.856,"output":14.246,"cache_read":1.428}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"upstage/solar-pro4:thinking":{"id":"upstage/solar-pro4:thinking","name":"Solar Pro 4 Thinking","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"input":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":80000},"cost":{"input":0.25,"output":0.9,"cache_read":0.125}},"tencent/hy3":{"id":"tencent/hy3","name":"Tencent Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":128000},"cost":{"input":0.066,"output":0.26,"cache_read":0.029}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Tencent Hy4 Preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"TheDrummer/skyfall-36b-v2":{"id":"TheDrummer/skyfall-36b-v2","name":"TheDrummer Skyfall 36B V2","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"TheDrummer/UnslopNemo-12B-v4.1":{"id":"TheDrummer/UnslopNemo-12B-v4.1","name":"UnslopNemo 12b v4","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"input":8192,"output":26214},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"TheDrummer/Cydonia-24B-v4.3":{"id":"TheDrummer/Cydonia-24B-v4.3","name":"The Drummer Cydonia 24B v4.3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.12,"output":0.15,"cache_read":0.06}},"TheDrummer/Artemis-v1.1":{"id":"TheDrummer/Artemis-v1.1","name":"TheDrummer/Artemis v1.1","description":"TheDrummer's Artemis v1.1 is a Gemma 4 31B fine-tune for creative writing, expressive dialogue, and roleplay, with optional thinking and a 262K context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-09-06","last_updated":"2026-09-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.1,"output":0.45,"cache_read":0.05}},"TheDrummer/Cydonia-24B-v2":{"id":"TheDrummer/Cydonia-24B-v2","name":"The Drummer Cydonia 24B v2","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"TheDrummer/Cydonia-24B-v4":{"id":"TheDrummer/Cydonia-24B-v4","name":"The Drummer Cydonia 24B v4","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.2006,"output":0.2414,"cache_read":0.1003}},"TheDrummer/Anubis-70B-v1.1":{"id":"TheDrummer/Anubis-70B-v1.1","name":"Anubis 70B v1.1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":16384},"cost":{"input":0.31,"output":0.31,"cache_read":0.155}},"TheDrummer/Magidonia-24B-v4.3":{"id":"TheDrummer/Magidonia-24B-v4.3","name":"The Drummer Magidonia 24B v4.3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-25","last_updated":"2025-12-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.1003,"output":0.1207,"cache_read":0.05015}},"TheDrummer/Cydonia-24B-v4.1":{"id":"TheDrummer/Cydonia-24B-v4.1","name":"The Drummer Cydonia 24B v4.1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":131072,"output":117964},"cost":{"input":0.35,"output":0.55,"cache_read":0.16}},"TheDrummer/Anubis-70B-v1":{"id":"TheDrummer/Anubis-70B-v1","name":"Anubis 70B v1","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.31,"output":0.31,"cache_read":0.155}},"TheDrummer/Rocinante-12B-v1.1":{"id":"TheDrummer/Rocinante-12B-v1.1","name":"Rocinante 12b","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":0.408,"output":0.595,"cache_read":0.204}},"soob3123/Veiled-Calla-12B":{"id":"soob3123/Veiled-Calla-12B","name":"Veiled Calla 12B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-04-13","last_updated":"2025-04-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"soob3123/amoral-gemma3-27B-v2":{"id":"soob3123/amoral-gemma3-27B-v2","name":"Amoral Gemma3 27B v2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-05-23","last_updated":"2025-05-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":8192},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"soob3123/GrayLine-Qwen3-8B":{"id":"soob3123/GrayLine-Qwen3-8B","name":"Grayline Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.3,"output":0.3,"cache_read":0.15}},"mistral/mistral-medium-3.5:thinking":{"id":"mistral/mistral-medium-3.5:thinking","name":"Mistral Medium 3.5 Thinking","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.5,"output":7.5,"cache_read":0.75}},"mistral/mistral-medium-3.5":{"id":"mistral/mistral-medium-3.5","name":"Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":32768},"cost":{"input":1.5,"output":7.5,"cache_read":0.75}},"nanogpt/coding-router:low":{"id":"nanogpt/coding-router:low","name":"Coding Router Low","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"nanogpt/coding-router:high":{"id":"nanogpt/coding-router:high","name":"Coding Router High","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"nanogpt/coding-router":{"id":"nanogpt/coding-router","name":"Coding Router","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":1.1,"output":2.2,"cache_read":0.11}},"nanogpt/coding-router:max":{"id":"nanogpt/coding-router:max","name":"Coding Router Max","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"nanogpt/coding-router:medium":{"id":"nanogpt/coding-router:medium","name":"Coding Router Medium","description":"Automatic model router for matching prompts to suitable backends and budgets","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"liquid/lfm-2.5-2.6b":{"id":"liquid/lfm-2.5-2.6b","name":"LFM2.5 2.6B","description":"Liquid AI's compact 2.6B reasoning model for agent workflows, data extraction, RAG, and long-context processing. It supports tool calling and structured output, but Liquid advises against using it for agentic coding. Warning: prompts and responses may be logged and used for model training or service improvement; do not send sensitive data.","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-4.5v:thinking":{"id":"z-ai/glm-4.5v:thinking","name":"GLM 4.5V Thinking","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.3}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":24000},"cost":{"input":0.3,"output":0.9,"cache_read":0.15}},"z-ai/glm-4.6v-original":{"id":"z-ai/glm-4.6v-original","name":"GLM 4.6V Original","description":"GLM-4.6V scales its context window to 128k tokens in training, and achieves SoTA performance in visual understanding among models of similar parameter scales. Integrates native Function Calling capabilities, bridging 'visual perception' and 'executable action' for multimodal agents. Direct via Z-AI (Zhipu).","family":"glm","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":24000},"cost":{"input":0.6,"output":0.9,"cache_read":0.3}},"z-ai/glm-5.3:thinking":{"id":"z-ai/glm-5.3:thinking","name":"GLM 5.3 Thinking","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/GLM-4.6-turbo":{"id":"z-ai/GLM-4.6-turbo","name":"GLM 4.6 Turbo","description":"Fast variant of GLM 4.6 for general chat, coding, and analysis with improved latency and strong reasoning.","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.5}},"z-ai/GLM-4.5-Air:thinking":{"id":"z-ai/GLM-4.5-Air:thinking","name":"GLM 4.5 Air (Thinking)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":98304},"cost":{"input":0.12,"output":0.8,"cache_read":0.06}},"z-ai/glm-4.6:thinking":{"id":"z-ai/glm-4.6:thinking","name":"GLM 4.6 Thinking","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.42,"output":1.32,"cache_read":0.078}},"z-ai/glm-4.7-original":{"id":"z-ai/glm-4.7-original","name":"GLM 4.7 Original","description":"GLM-4.7 is a next-gen GLM series text model with stronger reasoning, long-context chat, and reliable tool use. Routed directly via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.7-flash:thinking":{"id":"z-ai/glm-4.7-flash:thinking","name":"GLM 4.7 Flash Thinking","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"z-ai/glm-4.7:thinking":{"id":"z-ai/glm-4.7:thinking","name":"GLM 4.7 Thinking","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.2,"output":0.8,"cache_read":0.1}},"z-ai/glm-5v-turbo:thinking":{"id":"z-ai/glm-5v-turbo:thinking","name":"GLM 5V Turbo Thinking","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-flash-original":{"id":"z-ai/glm-4.7-flash-original","name":"GLM 4.7 Flash Original","description":"GLM-4.7-Flash is a lightweight 30B model optimized for coding and agentic tasks. Balances high performance with efficiency, perfect for local deployment.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"z-ai/GLM-4.5:thinking":{"id":"z-ai/GLM-4.5:thinking","name":"GLM 4.5 (Thinking)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.3,"cache_read":0.15}},"z-ai/glm-5-original:thinking":{"id":"z-ai/glm-5-original:thinking","name":"GLM 5 Original Thinking","description":"GLM-5 original with extended thinking capabilities for complex reasoning.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.3,"output":1.3,"cache_read":0.15}},"z-ai/GLM-4.5-Air":{"id":"z-ai/GLM-4.5-Air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":98304},"cost":{"input":0.12,"output":0.8,"cache_read":0.06}},"z-ai/glm-4.7-original:thinking":{"id":"z-ai/glm-4.7-original:thinking","name":"GLM 4.7 Original Thinking","description":"GLM-4.7 original with extended thinking capabilities for complex reasoning.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65535},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"input":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.3}},"z-ai/glm-5.1:thinking":{"id":"z-ai/glm-5.1:thinking","name":"GLM 5.1 Thinking","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.75,"output":2.6,"cache_read":0.15}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.5,"output":2.55,"cache_read":0.13}},"z-ai/GLM-4.6-turbo:thinking":{"id":"z-ai/GLM-4.6-turbo:thinking","name":"GLM 4.6 Turbo (Thinking)","description":"GLM 4.6 Turbo with thinking mode enabled for enhanced reasoning; shows internal reasoning and supports long context.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"input":204800,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.5}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":131072},"cost":{"input":0.75,"output":2.6,"cache_read":0.15}},"z-ai/glm-5.2:thinking":{"id":"z-ai/glm-5.2:thinking","name":"GLM 5.2 Thinking","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":0.42,"output":1.32,"cache_read":0.078}},"z-ai/glm-latest":{"id":"z-ai/glm-latest","name":"GLM Latest","description":"Compatibility alias that routes to the newest thinking GLM model. Currently routes to GLM 5.2 Thinking.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-05-03","last_updated":"2026-05-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-flash-original:thinking":{"id":"z-ai/glm-4.7-flash-original:thinking","name":"GLM 4.7 Flash Original Thinking","description":"GLM-4.7-Flash with extended thinking capabilities for complex reasoning. Lightweight 30B model optimized for coding and agentic tasks.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"z-ai/glm-5-original":{"id":"z-ai/glm-5-original","name":"GLM 5 Original","description":"GLM-5 is Zhipu's latest flagship model with advanced reasoning and instruction following. Routed directly via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"input":202800,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3-flash-uncensored":{"id":"z-ai/glm-5.3-flash-uncensored","name":"GLM 5.3 Flash Uncensored","description":"GLM 5.3 Flash Uncensored is an uncensored fine-tune of the efficient 320B mixture-of-experts reasoning model, built for unrestricted chat, creative writing, coding, agentic work, tool use, and long-context tasks.","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-29","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"input":1048576,"output":32768},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-4.6-original":{"id":"z-ai/glm-4.6-original","name":"GLM 4.6 Original","description":"GLM-4.6, Zhipu's flagship text model with 256K context window and advanced reasoning capabilities. Direct via Z-AI (Zhipu).","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":65535},"cost":{"input":0.35,"output":1.4,"cache_read":0.175}},"z-ai/glm-5:thinking":{"id":"z-ai/glm-5:thinking","name":"GLM 5 Thinking","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.5,"output":2.55,"cache_read":0.13}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"Doctor-Shotgun/MS3.2-24B-Magnum-Diamond":{"id":"Doctor-Shotgun/MS3.2-24B-Magnum-Diamond","name":"MS3.2 24B Magnum Diamond","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0.493,"output":0.493,"cache_read":0.2465}},"THUDM/GLM-4-9B-0414":{"id":"THUDM/GLM-4-9B-0414","name":"GLM 4 9B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8000},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"THUDM/GLM-Z1-9B-0414":{"id":"THUDM/GLM-Z1-9B-0414","name":"GLM Z1 9B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm-z","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"input":32000,"output":8000},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}},"THUDM/GLM-4-32B-0414":{"id":"THUDM/GLM-4-32B-0414","name":"GLM 4 32B 0414","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2024-01-01","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":128000,"output":65536},"cost":{"input":0.2,"output":0.2,"cache_read":0.1}}}},"watsonx":{"id":"watsonx","env":["WATSONX_AI_APIKEY","WATSONX_AI_PROJECT_ID"],"npm":"watsonx-ai-provider","name":"watsonx.ai","doc":"https://www.ibm.com/docs/en/watsonx/saas?topic=solutions-supported-foundation-models","models":{"mistralai/mistral-small-3-1-24b-instruct-2503":{"id":"mistralai/mistral-small-3-1-24b-instruct-2503","name":"Mistral Small 3.1 24B","description":"Efficient multimodal model for instruction following, coding, reasoning, and function calling","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.106,"output":0.318}},"ibm/granite-4-h-small":{"id":"ibm/granite-4-h-small","name":"Granite-4.0-H-Small","description":"Open-weight hybrid model for enterprise chat, coding, retrieval-augmented generation, and tool-calling workloads","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0636,"output":0.265}},"meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta-llama/llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.371,"output":1.484}},"meta-llama/llama-3-3-70b-instruct":{"id":"meta-llama/llama-3-3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.7526,"output":0.7526}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.159,"output":0.636}}}},"digitalocean":{"id":"digitalocean","env":["DIGITALOCEAN_ACCESS_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.do-ai.run/v1","name":"DigitalOcean","doc":"https://docs.digitalocean.com/products/gradient-ai-platform/details/models/","models":{"openai-gpt-4o":{"id":"openai-gpt-4o","name":"OpenAI GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai-gpt-5.2-pro":{"id":"openai-gpt-5.2-pro","name":"OpenAI GPT-5.2 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":21,"output":168}},"bge-reranker-v2-m3":{"id":"bge-reranker-v2-m3","name":"BGE Reranker v2 M3","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-12","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1},"cost":{"input":0.01,"output":0}},"anthropic-claude-opus-4.6":{"id":"anthropic-claude-opus-4.6","name":"Anthropic Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"openai-o3":{"id":"openai-o3","name":"OpenAI o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.08,"output":0.252,"cache_read":0.0252}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":131072}},"qwen-2.5-14b-instruct":{"id":"qwen-2.5-14b-instruct","name":"Qwen 2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072}},"nvidia-nemotron-3-super-120b":{"id":"nvidia-nemotron-3-super-120b","name":"NVIDIA Nemotron 3 Super 120B  (Public Preview)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.3,"output":0.65,"cache_read":0.06}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":1.7,"cache_read":0.09}},"openai-gpt-5.6-terra":{"id":"openai-gpt-5.6-terra","name":"OpenAI GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemma-4-31B-it":{"id":"gemma-4-31B-it","name":"Gemma 4","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.18,"output":0.5,"cache_read":0.036}},"alibaba-qwen3-32b":{"id":"alibaba-qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.25,"output":0.55}},"openai-gpt-image-1.5":{"id":"openai-gpt-image-1.5","name":"OpenAI GPT Image 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["image","text"]},"open_weights":false,"limit":{"context":0,"output":16384},"cost":{"input":5,"output":10,"cache_read":1}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"anthropic-claude-opus-4":{"id":"anthropic-claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"openai-gpt-6-astra":{"id":"openai-gpt-6-astra","name":"OpenAI GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"tiers":[{"input":20,"output":75,"cache_read":2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2}}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":2.2,"cache_read":0.105}},"arcee-trinity-large-thinking":{"id":"arcee-trinity-large-thinking","name":"Arcee Trinity Large Thinking (Public Preview)","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.25,"output":0.9,"cache_read":0.06}},"anthropic-claude-opus-4.7":{"id":"anthropic-claude-opus-4.7","name":"Anthropic Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic-claude-fable-5":{"id":"anthropic-claude-fable-5","name":"Anthropic Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic-claude-3.5-sonnet":{"id":"anthropic-claude-3.5-sonnet","name":"Claude 3.5 Sonnet","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-06-20","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax M2.5 (Public Preview)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-12","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"openai-gpt-4.1":{"id":"openai-gpt-4.1","name":"OpenAI GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai-gpt-5.3-codex":{"id":"openai-gpt-5.3-codex","name":"OpenAI GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai-gpt-oss-120b":{"id":"openai-gpt-oss-120b","name":"OpenAI GPT-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.055,"output":0.385,"cache_read":0.02}},"openai-gpt-5.2":{"id":"openai-gpt-5.2","name":"OpenAI GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"anthropic-claude-4.5-haiku":{"id":"anthropic-claude-4.5-haiku","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":1,"cache_write":1.25}},"e5-large-v2":{"id":"e5-large-v2","name":"E5 Large v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-05-19","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.02,"output":0}},"openai-gpt-5.6-luna":{"id":"openai-gpt-5.6-luna","name":"OpenAI GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04}}},"anthropic-claude-5-sonnet":{"id":"anthropic-claude-5-sonnet","name":"Anthropic Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai-gpt-oss-20b":{"id":"openai-gpt-oss-20b","name":"OpenAI GPT-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.45}},"deepseek-3.2":{"id":"deepseek-3.2","name":"Deepseek 3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-12-02","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":0.8,"cache_read":0.075}},"multi-qa-mpnet-base-dot-v1":{"id":"multi-qa-mpnet-base-dot-v1","name":"Multi-QA-mpnet-base-dot-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":768},"cost":{"input":0.009,"output":0}},"anthropic-claude-3.7-sonnet":{"id":"anthropic-claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"nemotron-3-nano-30b":{"id":"nemotron-3-nano-30b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"gte-large-en-v1.5":{"id":"gte-large-en-v1.5","name":"GTE Large (v1.5)","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-27","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0.09,"output":0}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen 3.5 397B A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.55,"output":3.5,"cache_read":0.111}},"openai-gpt-5":{"id":"openai-gpt-5","name":"OpenAI GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.696}},"llama3.3-70b-instruct":{"id":"llama3.3-70b-instruct","name":"Llama 3.3 Instruct (70B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.65,"output":0.65}},"all-mini-lm-l6-v2":{"id":"all-mini-lm-l6-v2","name":"All-MiniLM-L6-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256,"output":384},"cost":{"input":0.009,"output":0}},"anthropic-claude-sonnet-4":{"id":"anthropic-claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.3,"cache_write":3.75,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.3,"cache_write":3.75}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.55,"output":12.95,"cache_read":0.285}},"openai-gpt-5.4-mini":{"id":"openai-gpt-5.4-mini","name":"OpenAI GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"anthropic-claude-opus-4.8":{"id":"anthropic-claude-opus-4.8","name":"Anthropic Claude Opus 4.8","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai-gpt-5.4-pro":{"id":"openai-gpt-5.4-pro","name":"OpenAI GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"deepseek-4-flash":{"id":"deepseek-4-flash","name":"Deepseek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-05-27","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":384000},"cost":{"input":0.0679,"output":0.168,"cache_read":0.0168}},"openai-gpt-5.6-sol":{"id":"openai-gpt-5.6-sol","name":"OpenAI GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"tiers":[{"input":8,"output":30,"cache_read":0.8,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8}}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"Mistral Nemo Instruct","description":"Legacy model retained for compatibility with older integrations","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.3,"output":0.3}},"nemotron-3-ultra-550b":{"id":"nemotron-3-ultra-550b","name":"Nemotron 3 Ultra","description":"Flagship Nemotron model for high-throughput reasoning and complex agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.9,"output":1.7}},"anthropic-claude-fable-5.1":{"id":"anthropic-claude-fable-5.1","name":"Anthropic Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"nemotron-nano-12b-v2-vl":{"id":"nemotron-nano-12b-v2-vl","name":"Nemotron-nano 12b v2-vl","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.6}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32678,"output":8192},"cost":{"input":0.99,"output":0.99}},"openai-o3-mini":{"id":"openai-o3-mini","name":"OpenAI o3 mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai-gpt-5.1-codex-max":{"id":"openai-gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"nemotron-3-nano-omni":{"id":"nemotron-3-nano-omni","name":"Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":0.9}},"openai-gpt-5.4":{"id":"openai-gpt-5.4","name":"OpenAI GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai-gpt-5-nano":{"id":"openai-gpt-5-nano","name":"OpenAI GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"anthropic-claude-4.5-sonnet":{"id":"anthropic-claude-4.5-sonnet","name":"Anthropic Claude 4.5 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"Mistral 7B Instruct v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-05-22","last_updated":"2024-05-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768}},"ministral-3-8b-instruct-2512":{"id":"ministral-3-8b-instruct-2512","name":"Ministral 3 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai-gpt-5.5":{"id":"openai-gpt-5.5","name":"OpenAI GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"glm-5":{"id":"glm-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":64000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"openai-gpt-5-mini":{"id":"openai-gpt-5-mini","name":"OpenAI GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"anthropic-claude-3.5-haiku":{"id":"anthropic-claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-11-05","last_updated":"2024-11-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8-Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.7,"cache_read":0.203}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":1.3,"output":4.3,"cache_read":0.26}},"mistral-3-14B":{"id":"mistral-3-14B","name":"Ministral 3 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.2,"output":0.2}},"qwen3-tts-voicedesign":{"id":"qwen3-tts-voicedesign","name":"Qwen3 TTS VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":32768,"output":1}},"bge-m3":{"id":"bge-m3","name":"BGE M3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0.02,"output":0}},"qwen3-embedding-0.6b":{"id":"qwen3-embedding-0.6b","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":1024},"status":"beta","cost":{"input":0.04,"output":0}},"openai-o1":{"id":"openai-o1","name":"OpenAI o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"wan2-2-t2v-a14b":{"id":"wan2-2-t2v-a14b","name":"Wan2.2-T2V-A14B","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-07-28","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["video"]},"open_weights":true,"limit":{"context":100,"output":1},"cost":{"input":0.6,"output":0}},"anthropic-claude-3-opus":{"id":"anthropic-claude-3-opus","name":"Claude 3 Opus","description":"Legacy model retained for compatibility with older integrations","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08","release_date":"2024-02-29","last_updated":"2024-02-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic-claude-4.1-opus":{"id":"anthropic-claude-4.1-opus","name":"Anthropic Claude 4.1 Opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"openai-gpt-image-1":{"id":"openai-gpt-image-1","name":"GPT Image 1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"Deepseek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.87,"output":1.74,"cache_read":0.174}},"stable-diffusion-3.5-large":{"id":"stable-diffusion-3.5-large","name":"Stable Diffusion 3.5 Large","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-10-22","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":256,"output":1},"cost":{"input":0.08,"output":0}},"anthropic-claude-opus-4.5":{"id":"anthropic-claude-opus-4.5","name":"Anthropic Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"llama3-8b-instruct":{"id":"llama3-8b-instruct","name":"Llama 3.1 Instruct (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.198,"output":0.198}},"glm-5.3":{"id":"glm-5.3","name":"GLM5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"anthropic-claude-opus-5":{"id":"anthropic-claude-opus-5","name":"Anthropic Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai-gpt-5.4-nano":{"id":"openai-gpt-5.4-nano","name":"OpenAI GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"anthropic-claude-4.6-sonnet":{"id":"anthropic-claude-4.6-sonnet","name":"Anthropic Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic-claude-haiku-4.5":{"id":"anthropic-claude-haiku-4.5","name":"Anthropic Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":1.5,"cache_read":0.08}},"openai-gpt-image-2":{"id":"openai-gpt-image-2","name":"OpenAI GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image","text"]},"open_weights":false,"limit":{"context":0,"output":16384},"cost":{"input":8,"output":30}},"openai-gpt-4o-mini":{"id":"openai-gpt-4o-mini","name":"OpenAI GPT-4o mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"fal-ai/fast-sdxl":{"id":"fal-ai/fast-sdxl","name":"Fast SDXL","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-07-26","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":0,"output":0}},"fal-ai/elevenlabs/tts/multilingual-v2":{"id":"fal-ai/elevenlabs/tts/multilingual-v2","name":"ElevenLabs Multilingual TTS v2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-08-22","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fal-ai/stable-audio-25/text-to-audio":{"id":"fal-ai/stable-audio-25/text-to-audio","name":"Stable Audio 2.5 (Text-to-Audio)","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-08","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fal-ai/flux/schnell":{"id":"fal-ai/flux/schnell","name":"FLUX.1 [schnell]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-01","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":0,"output":0}}}},"vivgrid":{"id":"vivgrid","env":["VIVGRID_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.vivgrid.com/v1","name":"Vivgrid","doc":"https://docs.vivgrid.com/models","models":{"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.35,"output":3,"reasoning":3,"cache_read":0.05}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.2,"cache_read":0.3}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.3,"reasoning":0.3,"cache_read":0.03}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":10,"output":50,"cache_read":0.5,"cache_write":12.5}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.28,"output":0.42}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.04}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":10,"output":50,"cache_read":1.25,"cache_write":12.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":3.75,"cache_read":0.15}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"auriko":{"id":"auriko","env":["AURIKO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.auriko.ai/v1","name":"Auriko","doc":"https://docs.auriko.ai","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_write":0.375}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"qwen-3.6-plus":{"id":"qwen-3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_write":0.375}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}}}},"siliconflow-cn":{"id":"siliconflow-cn","env":["SILICONFLOW_CN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.siliconflow.cn/v1","name":"SiliconFlow (China)","doc":"https://cloud.siliconflow.com/models","models":{"baidu/ERNIE-4.5-300B-A47B":{"id":"baidu/ERNIE-4.5-300B-A47B","name":"baidu/ERNIE-4.5-300B-A47B","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-02","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.28,"output":1.1}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"stepfun-ai/Step-3.5-Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","family":"step","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.1,"output":0.3}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.003}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"deepseek-ai/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-OCR":{"id":"deepseek-ai/DeepSeek-OCR","name":"deepseek-ai/DeepSeek-OCR","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"deepseek-ai/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"deepseek-ai/DeepSeek-V4-Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":393000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"deepseek-ai/DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"inclusionAI/Ling-flash-2.0":{"id":"inclusionAI/Ling-flash-2.0","name":"inclusionAI/Ling-flash-2.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"zai-org/GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.86}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen/Qwen3.5-27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.09}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen/Qwen3-8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.06,"output":0.06}},"Qwen/Qwen3-14B":{"id":"Qwen/Qwen3-14B","name":"Qwen/Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3.5-4B":{"id":"Qwen/Qwen3.5-4B","name":"Qwen/Qwen3.5-4B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen/Qwen3.5-9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.22,"output":1.74}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen/Qwen3.5-122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.32}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen/Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.74}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen/Qwen3.5-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.23,"output":1.86}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen/Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen/Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen/Qwen3.6-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.23,"output":1.86}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen/Qwen3-VL-30B-A3B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen/Qwen3-VL-235B-A22B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-04","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":3.5}},"Qwen/Qwen3-VL-8B-Instruct":{"id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen/Qwen3-VL-8B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.18,"output":0.68}},"Qwen/Qwen3-VL-32B-Instruct":{"id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen/Qwen3-VL-32B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen/Qwen2.5-72B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.59,"output":0.59}},"Qwen/Qwen2.5-7B-Instruct":{"id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen/Qwen2.5-7B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.05,"output":0.05}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.25,"output":1}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen/Qwen3-VL-235B-A22B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-04","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.3,"output":1.5}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen/Qwen3-Coder-30B-A3B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3-VL-32B-Thinking":{"id":"Qwen/Qwen3-VL-32B-Thinking","name":"Qwen/Qwen3-VL-32B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-VL-30B-A3B-Thinking":{"id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen/Qwen3-VL-30B-A3B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.3}},"ByteDance-Seed/Seed-OSS-36B-Instruct":{"id":"ByteDance-Seed/Seed-OSS-36B-Instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.21,"output":0.57}},"Pro/deepseek-ai/DeepSeek-V3":{"id":"Pro/deepseek-ai/DeepSeek-V3","name":"Pro/deepseek-ai/DeepSeek-V3","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"Pro/deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"Pro/deepseek-ai/DeepSeek-V3.1-Terminus","name":"Pro/deepseek-ai/DeepSeek-V3.1-Terminus","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"Pro/deepseek-ai/DeepSeek-R1":{"id":"Pro/deepseek-ai/DeepSeek-R1","name":"Pro/deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"Pro/deepseek-ai/DeepSeek-V3.2":{"id":"Pro/deepseek-ai/DeepSeek-V3.2","name":"Pro/deepseek-ai/DeepSeek-V3.2","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"Pro/zai-org/GLM-5.1":{"id":"Pro/zai-org/GLM-5.1","name":"Pro/zai-org/GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"Pro/zai-org/GLM-5":{"id":"Pro/zai-org/GLM-5","name":"Pro/zai-org/GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1,"output":3.2}},"Pro/MiniMaxAI/MiniMax-M2.5":{"id":"Pro/MiniMaxAI/MiniMax-M2.5","name":"Pro/MiniMaxAI/MiniMax-M2.5","description":"Frontier MiniMax model for engineering, office tasks, and agentic reasoning","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":131000},"cost":{"input":0.3,"output":1.22}},"Pro/moonshotai/Kimi-K2.5":{"id":"Pro/moonshotai/Kimi-K2.5","name":"Pro/moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"Pro/moonshotai/Kimi-K2.6":{"id":"Pro/moonshotai/Kimi-K2.6","name":"Pro/moonshotai/Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"tencent/Hunyuan-A13B-Instruct":{"id":"tencent/Hunyuan-A13B-Instruct","name":"tencent/Hunyuan-A13B-Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"PaddlePaddle/PaddleOCR-VL-1.5":{"id":"PaddlePaddle/PaddleOCR-VL-1.5","name":"PaddlePaddle/PaddleOCR-VL-1.5","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-29","last_updated":"2026-01-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0,"output":0}}}},"nova":{"id":"nova","env":["NOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.nova.amazon.com/v1","name":"Nova","doc":"https://nova.amazon.com/dev/documentation","models":{"nova-2-lite-v1":{"id":"nova-2-lite-v1","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"reasoning":0}},"nova-2-pro-v1":{"id":"nova-2-pro-v1","name":"Nova 2 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2026-01-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"reasoning":0}}}},"inceptron":{"id":"inceptron","env":["INCEPTRON_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inceptron.io/v1","name":"Inceptron","doc":"https://docs.inceptron.io","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.13,"output":0.28,"cache_read":0.03,"cache_write":0}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.71,"output":2.35,"cache_read":0.12,"cache_write":0}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.66,"output":3.4,"cache_read":0.18,"cache_write":0}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.53,"output":3.39,"cache_read":0.17,"cache_write":0}}}},"vultr":{"id":"vultr","env":["VULTR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.vultrinference.com/v1","name":"Vultr","doc":"https://api.vultrinference.com/","models":{"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.3,"output":1}},"nvidia/DeepSeek-V3.2-NVFP4":{"id":"nvidia/DeepSeek-V3.2-NVFP4","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.55,"output":1.65}},"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16":{"id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning-BF16","name":"NVIDIA Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.13,"output":0.38}},"nvidia/Nemotron-Cascade-2-30B-A3B":{"id":"nvidia/Nemotron-Cascade-2-30B-A3B","name":"NVIDIA Nemotron Cascade 2","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":393216,"output":131072},"cost":{"input":0.85,"output":3.1}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":1.2}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.55,"output":1.65}}}},"ollama-cloud":{"id":"ollama-cloud","env":["OLLAMA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ollama.com/v1","name":"Ollama Cloud","doc":"https://docs.ollama.com/cloud","models":{"gpt-oss:20b":{"id":"gpt-oss:20b","name":"gpt-oss:20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768}},"deepseek-v4-flash:0731":{"id":"deepseek-v4-flash:0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576}},"minimax-m2.7":{"id":"minimax-m2.7","name":"minimax-m2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608}},"kimi-k2.6":{"id":"kimi-k2.6","name":"kimi-k2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":976000,"output":131072}},"minimax-m2.5":{"id":"minimax-m2.5","name":"minimax-m2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072}},"minimax-m3":{"id":"minimax-m3","name":"minimax-m3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":131072}},"qwen3.5:397b":{"id":"qwen3.5:397b","name":"qwen3.5:397b","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"release_date":"2026-02-15","last_updated":"2026-02-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"kimi-k2.7-code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"gpt-oss:120b":{"id":"gpt-oss:120b","name":"gpt-oss:120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768}},"nemotron-3-ultra":{"id":"nemotron-3-ultra","name":"nemotron-3-ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000}},"nemotron-3-nano:30b":{"id":"nemotron-3-nano:30b","name":"nemotron-3-nano:30b","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072}},"mistral-large-3:675b":{"id":"mistral-large-3:675b","name":"mistral-large-3:675b","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-12-02","last_updated":"2026-01-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"kimi-k3":{"id":"kimi-k3","name":"kimi-k3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072}},"gemma4:31b":{"id":"gemma4:31b","name":"gemma4:31b","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"kimi-k2.5":{"id":"kimi-k2.5","name":"kimi-k2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"glm-5.1":{"id":"glm-5.1","name":"glm-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"release_date":"2026-03-27","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"deepseek-v4-pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072}},"nemotron-3-super":{"id":"nemotron-3-super","name":"nemotron-3-super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536}}}},"freemodel":{"id":"freemodel","env":["FREEMODEL_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://cc.freemodel.dev/v1","name":"FreeModel","doc":"https://freemodel.dev","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.175,"cache_write":1.75}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.freemodel.dev/v1"},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}}}},"iflowcn":{"id":"iflowcn","env":["IFLOW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://apis.iflow.cn/v1","name":"iFlow","doc":"https://platform.iflow.cn/en/docs","models":{"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3-Coder-Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3-235B-A22B-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"qwen3-235b-a22b-instruct":{"id":"qwen3-235b-a22b-instruct","name":"Qwen3-235B-A22B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-235b":{"id":"qwen3-235b","name":"Qwen3-235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"kimi-k2-0905":{"id":"kimi-k2-0905","name":"Kimi-K2-0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL-Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"qwen3-max-preview":{"id":"qwen3-max-preview","name":"Qwen3-Max-Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0,"output":0}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3-Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"kimi-k2":{"id":"kimi-k2","name":"Kimi-K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0,"output":0}}}},"scx-ai":{"id":"scx-ai","env":["SCX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scx.ai/v1","name":"SCX.ai","doc":"https://platform.scx.ai/docs","models":{"Qwen3.8-Max":{"id":"Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":983616,"output":131072},"cost":{"input":1.815,"output":5.4461,"cache_read":0.17,"cache_write":2.5}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.55,"output":1.784,"cache_read":0.111}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.17,"output":0.55}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.48,"output":1.79,"cache_read":0.05}}}},"evroc":{"id":"evroc","env":["EVROC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://models.think.evroc.com/v1","name":"evroc","doc":"https://docs.evroc.com/products/think/overview.html","models":{"evroc/roc":{"id":"evroc/roc","name":"roc","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-06-06","last_updated":"2026-06-06","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":2.875,"output":11.516}},"mistralai/Voxtral-Small-24B-2507":{"id":"mistralai/Voxtral-Small-24B-2507","name":"Voxtral Small 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["audio","text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"mistralai/Mistral-Medium-3.5-128B":{"id":"mistralai/Mistral-Medium-3.5-128B","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.725,"output":6.9}},"nvidia/Llama-3.3-70B-Instruct-FP8":{"id":"nvidia/Llama-3.3-70B-Instruct-FP8","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":1.15,"output":1.15}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.144,"output":0.575}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":1.4375,"output":5.75}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8-27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.87,"output":3.5}},"Qwen/Qwen3-Reranker-4B":{"id":"Qwen/Qwen3-Reranker-4B","name":"Qwen3 Reranker 4B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.0575,"output":0}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.345,"output":1.38}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":4096},"cost":{"input":0.115,"output":0.115}},"intfloat/multilingual-e5-large-instruct":{"id":"intfloat/multilingual-e5-large-instruct","name":"E5 Multi-Lingual Large Embeddings 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"cost":{"input":0.114,"output":0.114}},"KBLab/kb-whisper-large":{"id":"KBLab/kb-whisper-large","name":"KB Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper 3 Large","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/whisper-large-v3-turbo":{"id":"openai/whisper-large-v3-turbo","name":"Whisper Large v3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.0023,"output":0.0023,"output_audio":2.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.23,"output":0.92}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.4375,"output":5.75}}}},"echo":{"id":"echo","env":["ECHO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://echo.tracerml.ai/v1","name":"Echo","doc":"https://echo.tracerml.ai/docs/api","models":{"echo":{"id":"echo","name":"Echo","description":"Adaptive model for coding, reasoning, and tool-driven agent workflows through one OpenAI-compatible endpoint","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"beta","cost":{"input":10,"output":50}}}},"aixy":{"id":"aixy","env":["AIXY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.aixy-gateway.com/v1","name":"Aixy","doc":"https://docs.aixy-gateway.com/integrations/overview","models":{"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}}}},"impossibl":{"id":"impossibl","env":["IMPOSSIBL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.impossibl.com/v1","name":"Impossibl","doc":"https://impossibl.com/docs/models","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5}},"qwen/qwen3.8-max-preview":{"id":"qwen/qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"tiers":[{"input":1,"output":4,"cache_read":0.2,"tier":{"type":"context","size":262144}}],"context_over_200k":{"input":1,"output":4,"cache_read":0.2}}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":262144}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"groq/gpt-oss-20b":{"id":"groq/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/gpt-oss-120b":{"id":"groq/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.003}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.19,"output":0.51,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":3}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":3}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"fireworks/glm-5.2":{"id":"fireworks/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"fireworks/gpt-oss-20b":{"id":"fireworks/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.3,"cache_read":0.035}},"fireworks/gpt-oss-120b":{"id":"fireworks/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75}}}},"llmgateway-providers":{"id":"llmgateway-providers","env":["LLMGATEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmgateway.io/v1","name":"LLM Gateway","doc":"https://llmgateway.io/docs","models":{"vertex-openai/glm-4.7":{"id":"vertex-openai/glm-4.7","name":"GLM-4.7 (Vertex AI (OpenAI-compatible))","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.6,"output":2.2}},"vertex-openai/qwen3-next-80b-a3b-thinking":{"id":"vertex-openai/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking (Vertex AI (OpenAI-compatible))","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"vertex-openai/qwen3-next-80b-a3b-instruct":{"id":"vertex-openai/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct (Vertex AI (OpenAI-compatible))","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"vertex-openai/kimi-k2-thinking":{"id":"vertex-openai/kimi-k2-thinking","name":"Kimi K2 Thinking (Vertex AI (OpenAI-compatible))","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"vertex-openai/deepseek-v3.2":{"id":"vertex-openai/deepseek-v3.2","name":"DeepSeek V3.2 (Vertex AI (OpenAI-compatible))","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.56,"output":1.68,"cache_read":0.056}},"vertex-openai/glm-5":{"id":"vertex-openai/glm-5","name":"GLM-5 (Vertex AI (OpenAI-compatible))","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"cost":{"input":1,"output":3.2,"cache_read":0.1}},"vertex-openai/qwen3-235b-a22b-instruct-2507":{"id":"vertex-openai/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (Vertex AI (OpenAI-compatible))","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.22,"output":0.88}},"vertex-openai/grok-4-6":{"id":"vertex-openai/grok-4-6","name":"Grok 4.6 (Vertex AI (OpenAI-compatible))","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"vertex-openai/qwen3-coder-480b-a35b-instruct":{"id":"vertex-openai/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct (Vertex AI (OpenAI-compatible))","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"vertex-openai/grok-4-20-non-reasoning":{"id":"vertex-openai/grok-4-20-non-reasoning","name":"Grok 4.20 Non-Reasoning (Vertex AI (OpenAI-compatible))","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"vertex-openai/grok-4-20-reasoning":{"id":"vertex-openai/grok-4-20-reasoning","name":"Grok 4.20 Reasoning (Vertex AI (OpenAI-compatible))","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"baidu/kimi-k2.6":{"id":"baidu/kimi-k2.6","name":"Kimi K2.6 (Baidu)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"baidu/glm-5.2":{"id":"baidu/glm-5.2","name":"GLM-5.2 (Baidu)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/deepseek-v4-flash":{"id":"baidu/deepseek-v4-flash","name":"DeepSeek V4 Flash (Baidu)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":1.32,"cache_read":0.044}},"baidu/glm-5":{"id":"baidu/glm-5","name":"GLM-5 (Baidu)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"baidu/glm-5.1":{"id":"baidu/glm-5.1","name":"GLM-5.1 (Baidu)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"baidu/deepseek-v4-pro":{"id":"baidu/deepseek-v4-pro","name":"DeepSeek V4 Pro (Baidu)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.132}},"baidu/glm-5.3":{"id":"baidu/glm-5.3","name":"GLM-5.3 (Baidu)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"aws-mantle/gpt-5.6-sol":{"id":"aws-mantle/gpt-5.6-sol","name":"GPT-5.6 Sol (AWS Mantle)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":278528,"output":128000},"cost":{"input":5.5,"output":33,"cache_read":0.55,"cache_write":6.875}},"aws-mantle/gpt-5.6-luna":{"id":"aws-mantle/gpt-5.6-luna","name":"GPT-5.6 Luna (AWS Mantle)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":278528,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275}},"aws-mantle/gpt-5.6-terra":{"id":"aws-mantle/gpt-5.6-terra","name":"GPT-5.6 Terra (AWS Mantle)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":278528,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75}},"gonka24/minimax-m2.7":{"id":"gonka24/minimax-m2.7","name":"MiniMax M2.7 (Gonka24)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.08,"output":0.32,"cache_read":0.017}},"gonka24/deepseek-v4-flash":{"id":"gonka24/deepseek-v4-flash","name":"DeepSeek V4 Flash (Gonka24)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":390000,"output":16384},"cost":{"input":0.051,"output":0.104,"cache_read":0.0097}},"embercloud/glm-4.7":{"id":"embercloud/glm-4.7","name":"GLM-4.7 (EmberCloud)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.38,"output":1.98,"cache_read":0.19}},"embercloud/glm-4.5-air":{"id":"embercloud/glm-4.5-air","name":"GLM-4.5 Air (EmberCloud)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":96000},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"embercloud/glm-5.2":{"id":"embercloud/glm-5.2","name":"GLM-5.2 (EmberCloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"embercloud/qwen3-coder-next":{"id":"embercloud/qwen3-coder-next","name":"Qwen3 Coder Next (EmberCloud)","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.108,"output":0.675,"cache_read":0.06}},"embercloud/glm-4.5":{"id":"embercloud/glm-4.5","name":"GLM-4.5 (EmberCloud)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"embercloud/glm-5":{"id":"embercloud/glm-5","name":"GLM-5 (EmberCloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":0.72,"output":2.3,"cache_read":0.144}},"embercloud/kimi-k2.5":{"id":"embercloud/kimi-k2.5","name":"Kimi K2.5 (EmberCloud)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.405,"output":1.98,"cache_read":0.225}},"embercloud/glm-5.1":{"id":"embercloud/glm-5.1","name":"GLM-5.1 (EmberCloud)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131000},"cost":{"input":0.931,"output":2.93,"cache_read":0.173}},"embercloud/glm-4.7-flash":{"id":"embercloud/glm-4.7-flash","name":"GLM-4.7 Flash (EmberCloud)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"scx-ai/minimax-m2.7":{"id":"scx-ai/minimax-m2.7","name":"MiniMax M2.7 (SCX.ai (Turbo))","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.48,"output":1.79,"cache_read":0.05}},"scx-ai/qwen3-32b":{"id":"scx-ai/qwen3-32b","name":"Qwen3 32B (SCX.ai (Turbo))","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.36,"output":0.87}},"scx-ai/llama-4-maverick-17b-instruct":{"id":"scx-ai/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (SCX.ai (Turbo))","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.53,"output":1.62}},"scx-ai/gemma-4-31b-it":{"id":"scx-ai/gemma-4-31b-it","name":"Gemma 4 31B IT (SCX.ai (Turbo))","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.3,"output":0.91}},"scx-ai/gpt-oss-120b":{"id":"scx-ai/gpt-oss-120b","name":"GPT OSS 120B (SCX.ai (Turbo))","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.17,"output":0.55}},"groq/gpt-oss-20b":{"id":"groq/gpt-oss-20b","name":"GPT OSS 20B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32766},"cost":{"input":0.1,"output":0.5}},"groq/gpt-oss-120b":{"id":"groq/gpt-oss-120b","name":"GPT OSS 120B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32766},"cost":{"input":0.15,"output":0.75}},"google-vertex/gemini-3.1-pro-preview":{"id":"google-vertex/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Google Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google-vertex/gemini-2.5-flash-lite":{"id":"google-vertex/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite (Google Vertex AI)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google-vertex/gemini-3.6-flash":{"id":"google-vertex/gemini-3.6-flash","name":"Gemini 3.6 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-3.1-flash-lite":{"id":"google-vertex/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Google Vertex AI)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"google-vertex/gemini-3.5-flash":{"id":"google-vertex/gemini-3.5-flash","name":"Gemini 3.5 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"google-vertex/gemini-3.5-flash-lite":{"id":"google-vertex/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google-vertex/gemini-3-flash-preview":{"id":"google-vertex/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview) (Google Vertex AI)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google-vertex/gemini-3.8-flash":{"id":"google-vertex/gemini-3.8-flash","name":"Gemini 3.8 Flash (Google Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-3.7-flash":{"id":"google-vertex/gemini-3.7-flash","name":"Gemini 3.7 Flash (Google Vertex AI)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-vertex/gemini-2.5-pro":{"id":"google-vertex/gemini-2.5-pro","name":"Gemini 2.5 Pro (Google Vertex AI)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google-vertex/gemini-2.5-flash":{"id":"google-vertex/gemini-2.5-flash","name":"Gemini 2.5 Flash (Google Vertex AI)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"quartz/gemini-3.1-pro-preview":{"id":"quartz/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Quartz)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"vertex-anthropic/claude-sonnet-4-6":{"id":"vertex-anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Vertex AI (Anthropic))","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"vertex-anthropic/claude-opus-4-6":{"id":"vertex-anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Vertex AI (Anthropic))","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-anthropic/claude-opus-4-7":{"id":"vertex-anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Vertex AI (Anthropic))","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"vertex-anthropic/claude-haiku-4-5":{"id":"vertex-anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (Vertex AI (Anthropic))","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"vertex-anthropic/claude-sonnet-4-5":{"id":"vertex-anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (Vertex AI (Anthropic))","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"vertex-anthropic/claude-sonnet-5":{"id":"vertex-anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Vertex AI (Anthropic))","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"vertex-anthropic/claude-opus-4-5-20251101":{"id":"vertex-anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (Vertex AI (Anthropic))","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo V2.5 (Xiaomi)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro (Xiaomi)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning (MiniMax)","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":131072},"cost":{"input":0.12,"output":0.48}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1 (MiniMax)","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.27,"output":1.1}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2 (MiniMax)","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.2,"output":1,"cache_read":0.03}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 Highspeed (MiniMax)","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.06}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7 (MiniMax)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5 (MiniMax)","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3 (MiniMax)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed (MiniMax)","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.03}},"minimax/minimax-text-01":{"id":"minimax/minimax-text-01","name":"MiniMax Text 01 (MiniMax)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.2,"output":1.1}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen3.7 Max (Alibaba Cloud)","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"alibaba/qwen3-coder-plus":{"id":"alibaba/qwen3-coder-plus","name":"Qwen3 Coder Plus (Alibaba Cloud)","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":66000},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"alibaba/qwen35-397b-a17b":{"id":"alibaba/qwen35-397b-a17b","name":"Qwen3.5 397B A17B (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen3-coder-flash":{"id":"alibaba/qwen3-coder-flash","name":"Qwen3 Coder Flash (Alibaba Cloud)","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"alibaba/qwen-max":{"id":"alibaba/qwen-max","name":"Qwen Max (Alibaba Cloud)","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"alibaba/qwen3.6-plus":{"id":"alibaba/qwen3.6-plus","name":"Qwen3.6 Plus (Alibaba Cloud)","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"alibaba/qwen-flash":{"id":"alibaba/qwen-flash","name":"Qwen Flash (Alibaba Cloud)","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01,"cache_write":0.0625}},"alibaba/glm-5.2":{"id":"alibaba/glm-5.2","name":"GLM-5.2 (Alibaba Cloud)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28}},"alibaba/deepseek-v4-flash":{"id":"alibaba/deepseek-v4-flash","name":"DeepSeek V4 Flash (Alibaba Cloud)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"alibaba/qwen3-vl-plus":{"id":"alibaba/qwen3-vl-plus","name":"Qwen3 VL Plus (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"cache_read":0.04,"cache_write":0.25}},"alibaba/qwen-coder-plus":{"id":"alibaba/qwen-coder-plus","name":"Qwen Coder Plus (Alibaba Cloud)","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.502,"output":1.004}},"alibaba/qwen3.7-flash":{"id":"alibaba/qwen3.7-flash","name":"Qwen3.7 Flash (Alibaba Cloud)","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.0375}},"alibaba/qwen3.6-35b-a3b":{"id":"alibaba/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B (Alibaba Cloud)","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.375,"output":2.25}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max (Alibaba Cloud)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32800},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"alibaba/qwen3-vl-flash":{"id":"alibaba/qwen3-vl-flash","name":"Qwen3 VL Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"alibaba/qwen-plus":{"id":"alibaba/qwen-plus","name":"Qwen Plus (Alibaba Cloud)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32000},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"alibaba/qwen3.6-flash":{"id":"alibaba/qwen3.6-flash","name":"Qwen3.6 Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"cache_write":0.3125}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen3.8 Flash (Alibaba Cloud)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"alibaba/qwen3.6-max-preview":{"id":"alibaba/qwen3.6-max-preview","name":"Qwen3.6 Max Preview (Alibaba Cloud)","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13}},"alibaba/glm-5":{"id":"alibaba/glm-5","name":"GLM-5 (Alibaba Cloud)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0.573,"output":2.58}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen3.8 Max (Alibaba Cloud)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/kimi-k2.5":{"id":"alibaba/kimi-k2.5","name":"Kimi K2.5 (Alibaba Cloud)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0.574,"output":3.011}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen3.7 Plus (Alibaba Cloud)","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"alibaba/qwen-omni-turbo":{"id":"alibaba/qwen-omni-turbo","name":"Qwen Omni Turbo (Alibaba Cloud)","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.2,"output":0.8}},"alibaba/deepseek-v4-pro":{"id":"alibaba/deepseek-v4-pro","name":"DeepSeek V4 Pro (Alibaba Cloud)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":2.4,"output":4.8,"cache_read":0.2}},"alibaba/qwen-plus-latest":{"id":"alibaba/qwen-plus-latest","name":"Qwen Plus Latest (Alibaba Cloud)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-09-09","last_updated":"2024-09-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"scx-ai-gp/glm-5.2":{"id":"scx-ai-gp/glm-5.2","name":"GLM-5.2 (SCX.ai)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"scx-ai-gp/kimi-k2.7-code":{"id":"scx-ai-gp/kimi-k2.7-code","name":"Kimi K2.7 Code (SCX.ai)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"scx-ai-gp/kimi-k3":{"id":"scx-ai-gp/kimi-k3","name":"Kimi K3 (SCX.ai)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3.5,"output":18,"cache_read":0.35}},"scx-ai-gp/glm-5.2-fast":{"id":"scx-ai-gp/glm-5.2-fast","name":"GLM-5.2 Turbo (SCX.ai)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.2,"output":6.5,"cache_read":0.45}},"scx-ai-gp/glm-5.3-flash":{"id":"scx-ai-gp/glm-5.3-flash","name":"GLM-5.3 Flash (SCX.ai)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.088,"output":0.25,"cache_read":0.025}},"scx-ai-gp/qwen3.8-max":{"id":"scx-ai-gp/qwen3.8-max","name":"Qwen3.8 Max (SCX.ai)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"scx-ai-gp/glm-5.3":{"id":"scx-ai-gp/glm-5.3","name":"GLM-5.3 (SCX.ai)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"aws-bedrock/claude-sonnet-4-6":{"id":"aws-bedrock/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (AWS Bedrock)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/llama-4-scout-17b-instruct":{"id":"aws-bedrock/llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct (AWS Bedrock)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.17,"output":0.66}},"aws-bedrock/claude-opus-5":{"id":"aws-bedrock/claude-opus-5","name":"Claude Opus 5 (AWS Bedrock)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-opus-4-1-20250805":{"id":"aws-bedrock/claude-opus-4-1-20250805","name":"Claude Opus 4.1 (AWS Bedrock)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"aws-bedrock/claude-fable-5-1":{"id":"aws-bedrock/claude-fable-5-1","name":"Claude Fable 5.1 (AWS Bedrock)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"aws-bedrock/claude-opus-4-6":{"id":"aws-bedrock/claude-opus-4-6","name":"Claude Opus 4.6 (AWS Bedrock)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-sonnet-4-5-20250929":{"id":"aws-bedrock/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (2025-09-29) (AWS Bedrock)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/claude-opus-4-7":{"id":"aws-bedrock/claude-opus-4-7","name":"Claude Opus 4.7 (AWS Bedrock)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-haiku-4-5-20251001":{"id":"aws-bedrock/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (2025-10-01) (AWS Bedrock)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"aws-bedrock/claude-fable-5":{"id":"aws-bedrock/claude-fable-5","name":"Claude Fable 5 (AWS Bedrock)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"aws-bedrock/llama-4-maverick-17b-instruct":{"id":"aws-bedrock/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (AWS Bedrock)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.24,"output":0.97}},"aws-bedrock/grok-4-3":{"id":"aws-bedrock/grok-4-3","name":"Grok 4.3 (AWS Bedrock)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"aws-bedrock/claude-haiku-4-5":{"id":"aws-bedrock/claude-haiku-4-5","name":"Claude Haiku 4.5 (AWS Bedrock)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"aws-bedrock/claude-sonnet-4-5":{"id":"aws-bedrock/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (AWS Bedrock)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"aws-bedrock/llama-3.1-70b-instruct":{"id":"aws-bedrock/llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct (AWS Bedrock)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":2048},"cost":{"input":0.72,"output":0.72}},"aws-bedrock/grok-4-6":{"id":"aws-bedrock/grok-4-6","name":"Grok 4.6 (AWS Bedrock)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"aws-bedrock/claude-opus-4-8":{"id":"aws-bedrock/claude-opus-4-8","name":"Claude Opus 4.8 (AWS Bedrock)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"aws-bedrock/claude-sonnet-5":{"id":"aws-bedrock/claude-sonnet-5","name":"Claude Sonnet 5 (AWS Bedrock)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"aws-bedrock/claude-opus-4-5-20251101":{"id":"aws-bedrock/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (AWS Bedrock)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Anthropic)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5 (Anthropic)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1 (Anthropic)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Anthropic)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (2025-09-29) (Anthropic)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Anthropic)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (2025-10-01) (Anthropic)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5 (Anthropic)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (Anthropic)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (Anthropic)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8 (Anthropic)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Anthropic)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (Anthropic)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"canopywave/kimi-k2.6":{"id":"canopywave/kimi-k2.6","name":"Kimi K2.6 (CanopyWave)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"canopywave/glm-5.2":{"id":"canopywave/glm-5.2","name":"GLM-5.2 (CanopyWave)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"canopywave/deepseek-v4-flash":{"id":"canopywave/deepseek-v4-flash","name":"DeepSeek V4 Flash (CanopyWave)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"canopywave/kimi-k3":{"id":"canopywave/kimi-k3","name":"Kimi K3 (CanopyWave)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"canopywave/deepseek-v4-pro":{"id":"canopywave/deepseek-v4-pro","name":"DeepSeek V4 Pro (CanopyWave)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.74,"output":3.48,"cache_read":0.01}},"vichar-ai/glm-5.3-flash":{"id":"vichar-ai/glm-5.3-flash","name":"GLM-5.3 Flash (vichar-ai)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":128000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"vichar-ai/glm-5.3":{"id":"vichar-ai/glm-5.3","name":"GLM-5.3 (vichar-ai)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"together-ai/glm-4.7":{"id":"together-ai/glm-4.7","name":"GLM-4.7 (Together AI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.45,"output":2}},"together-ai/minimax-m3":{"id":"together-ai/minimax-m3","name":"MiniMax M3 (Together AI)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"together-ai/deepseek-v4-flash":{"id":"together-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash (Together AI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"together-ai/gpt-oss-20b":{"id":"together-ai/gpt-oss-20b","name":"GPT OSS 20B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2}},"together-ai/kimi-k3":{"id":"together-ai/kimi-k3","name":"Kimi K3 (Together AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":3,"output":15,"cache_read":0.3}},"together-ai/gemma-4-31b-it":{"id":"together-ai/gemma-4-31b-it","name":"Gemma 4 31B IT (Together AI)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.39,"output":0.97}},"together-ai/deepseek-v4-pro":{"id":"together-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro (Together AI)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":163840},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"together-ai/gpt-oss-120b":{"id":"together-ai/gpt-oss-120b","name":"GPT OSS 120B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3 (Meta)","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2 (Meta)","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1 (Meta)","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"google-ai-studio/gemini-pro-latest":{"id":"google-ai-studio/gemini-pro-latest","name":"Gemini Pro Latest (Google AI Studio)","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"google-ai-studio/gemini-3.1-pro-preview":{"id":"google-ai-studio/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro (Preview) (Google AI Studio)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google-ai-studio/gemini-2.5-flash-lite":{"id":"google-ai-studio/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite (Google AI Studio)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google-ai-studio/gemini-3.6-flash":{"id":"google-ai-studio/gemini-3.6-flash","name":"Gemini 3.6 Flash (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-3.1-flash-lite":{"id":"google-ai-studio/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Google AI Studio)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"google-ai-studio/gemini-3.5-flash":{"id":"google-ai-studio/gemini-3.5-flash","name":"Gemini 3.5 Flash (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"google-ai-studio/gemini-3.5-flash-lite":{"id":"google-ai-studio/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Google AI Studio)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"google-ai-studio/gemini-3-flash-preview":{"id":"google-ai-studio/gemini-3-flash-preview","name":"Gemini 3 Flash (Preview) (Google AI Studio)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google-ai-studio/gemini-3.8-flash":{"id":"google-ai-studio/gemini-3.8-flash","name":"Gemini 3.8 Flash (Google AI Studio)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-3.7-flash":{"id":"google-ai-studio/gemini-3.7-flash","name":"Gemini 3.7 Flash (Google AI Studio)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"google-ai-studio/gemini-2.5-pro":{"id":"google-ai-studio/gemini-2.5-pro","name":"Gemini 2.5 Pro (Google AI Studio)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google-ai-studio/gemini-2.5-flash":{"id":"google-ai-studio/gemini-2.5-flash","name":"Gemini 2.5 Flash (Google AI Studio)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"bytedance/glm-4.7":{"id":"bytedance/glm-4.7","name":"GLM-4.7 (ByteDance)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"bytedance/seed-1-8-251228":{"id":"bytedance/seed-1-8-251228","name":"Seed 1.8 (251228) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/glm-5.2":{"id":"bytedance/glm-5.2","name":"GLM-5.2 (ByteDance)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"bytedance/deepseek-v4-flash":{"id":"bytedance/deepseek-v4-flash","name":"DeepSeek V4 Flash (ByteDance)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"bytedance/seed-1-6-flash-250715":{"id":"bytedance/seed-1-6-flash-250715","name":"Seed 1.6 Flash (250715) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.07,"output":0.3,"cache_read":0.015}},"bytedance/deepseek-v3.2":{"id":"bytedance/deepseek-v3.2","name":"DeepSeek V3.2 (ByteDance)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.28,"output":0.42,"cache_read":0.056}},"bytedance/seed-1-6-250615":{"id":"bytedance/seed-1-6-250615","name":"Seed 1.6 (250615) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/deepseek-v4-pro":{"id":"bytedance/deepseek-v4-pro","name":"DeepSeek V4 Pro (ByteDance)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"bytedance/gpt-oss-120b":{"id":"bytedance/gpt-oss-120b","name":"GPT OSS 120B (ByteDance)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.1,"output":0.5,"cache_read":0.02}},"bytedance/seed-1-6-250915":{"id":"bytedance/seed-1-6-250915","name":"Seed 1.6 (250915) (ByteDance)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"novita/glm-4.7":{"id":"novita/glm-4.7","name":"GLM-4.7 (NovitaAI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"novita/qwen3.7-max":{"id":"novita/qwen3.7-max","name":"Qwen3.7 Max (NovitaAI)","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25}},"novita/gemma-4-26b-a4b-it":{"id":"novita/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT (NovitaAI)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.4}},"novita/llama-4-scout-17b-instruct":{"id":"novita/llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct (NovitaAI)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.18,"output":0.59}},"novita/qwen35-397b-a17b":{"id":"novita/qwen35-397b-a17b","name":"Qwen3.5 397B A17B (NovitaAI)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":64000},"cost":{"input":0.6,"output":3.6}},"novita/qwen3-235b-a22b-thinking-2507":{"id":"novita/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507 (NovitaAI)","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":3}},"novita/glm-4.6":{"id":"novita/glm-4.6","name":"GLM-4.6 (NovitaAI)","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11}},"novita/minimax-m2.1":{"id":"novita/minimax-m2.1","name":"MiniMax M2.1 (NovitaAI)","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"novita/glm-4.6v":{"id":"novita/glm-4.6v","name":"GLM-4.6V (NovitaAI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16000},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"novita/qwen3-next-80b-a3b-instruct":{"id":"novita/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct (NovitaAI)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"novita/ling-3.0-flash":{"id":"novita/ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash (NovitaAI)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"novita/qwen3.8-27b":{"id":"novita/qwen3.8-27b","name":"Qwen3.8 27B (NovitaAI)","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.42,"output":3,"cache_read":0.085}},"novita/qwen3-235b-a22b-fp8":{"id":"novita/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B FP8 (NovitaAI)","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"novita/minimax-m2.7":{"id":"novita/minimax-m2.7","name":"MiniMax M2.7 (NovitaAI)","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"novita/kimi-k2.6":{"id":"novita/kimi-k2.6","name":"Kimi K2.6 (NovitaAI)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"novita/glm-5.2":{"id":"novita/glm-5.2","name":"GLM-5.2 (NovitaAI)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"novita/minimax-m2.5":{"id":"novita/minimax-m2.5","name":"MiniMax M2.5 (NovitaAI)","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"novita/deepseek-v4-flash":{"id":"novita/deepseek-v4-flash","name":"DeepSeek V4 Flash (NovitaAI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"novita/kimi-k2.7-code":{"id":"novita/kimi-k2.7-code","name":"Kimi K2.7 Code (NovitaAI)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"novita/llama-3.2-3b-instruct":{"id":"novita/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct (NovitaAI)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"novita/hy3":{"id":"novita/hy3","name":"Hy3 (NovitaAI)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"novita/qwen3-coder-30b-a3b-instruct":{"id":"novita/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct (NovitaAI)","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.07,"output":0.27}},"novita/ernie-4.5-vl-424b-a47b":{"id":"novita/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B (NovitaAI)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"novita/kimi-k3":{"id":"novita/kimi-k3","name":"Kimi K3 (NovitaAI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"novita/deepseek-v3.2":{"id":"novita/deepseek-v3.2","name":"DeepSeek V3.2 (NovitaAI)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"novita/qwen3.6-35b-a3b":{"id":"novita/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B (NovitaAI)","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.248,"output":1.485}},"novita/qwen3-max":{"id":"novita/qwen3-max","name":"Qwen3 Max (NovitaAI)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.845,"output":3.38}},"novita/glm-5.3-flash":{"id":"novita/glm-5.3-flash","name":"GLM-5.3 Flash (NovitaAI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"novita/qwen3-vl-30b-a3b-instruct":{"id":"novita/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct (NovitaAI)","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-10-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.7}},"novita/llama-4-maverick-17b-instruct":{"id":"novita/llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct (NovitaAI)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.27,"output":0.85}},"novita/glm-4.5v":{"id":"novita/glm-4.5v","name":"GLM-4.5V (NovitaAI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"novita/qwen3.8-flash":{"id":"novita/qwen3.8-flash","name":"Qwen3.8 Flash (NovitaAI)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"novita/kimi-k2":{"id":"novita/kimi-k2","name":"Kimi K2 (NovitaAI)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"novita/gemma-4-31b-it":{"id":"novita/gemma-4-31b-it","name":"Gemma 4 31B IT (NovitaAI)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"novita/glm-5":{"id":"novita/glm-5","name":"GLM-5 (NovitaAI)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"novita/qwen3.8-max":{"id":"novita/qwen3.8-max","name":"Qwen3.8 Max (NovitaAI)","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"novita/glm-5.1":{"id":"novita/glm-5.1","name":"GLM-5.1 (NovitaAI)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.38,"output":4.4,"cache_read":0.26}},"novita/qwen3-vl-235b-a22b-thinking":{"id":"novita/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking (NovitaAI)","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"novita/qwen3-vl-235b-a22b-instruct":{"id":"novita/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct (NovitaAI)","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"novita/qwen3-235b-a22b-instruct-2507":{"id":"novita/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (NovitaAI)","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.58}},"novita/qwen3-coder-480b-a35b-instruct":{"id":"novita/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct (NovitaAI)","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"novita/glm-5.3":{"id":"novita/glm-5.3","name":"GLM-5.3 (NovitaAI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"novita/llama-3.3-70b-instruct":{"id":"novita/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct (NovitaAI)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":120000},"cost":{"input":0.135,"output":0.4}},"novita/llama-3-70b-instruct":{"id":"novita/llama-3-70b-instruct","name":"Llama 3 70B Instruct (NovitaAI)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"novita/mimo-v2.5":{"id":"novita/mimo-v2.5","name":"MiMo V2.5 (NovitaAI)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.168,"output":0.336,"cache_read":0.0034,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"novita/mimo-v2.5-pro":{"id":"novita/mimo-v2.5-pro","name":"MiMo V2.5 Pro (NovitaAI)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.522,"output":1.044,"cache_read":0.0043,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"ranoai/deepseek-v4-flash":{"id":"ranoai/deepseek-v4-flash","name":"DeepSeek V4 Flash (RanoAI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"inference.net/llama-3.2-11b-instruct":{"id":"inference.net/llama-3.2-11b-instruct","name":"Llama 3.2 11B Instruct (Inference.net)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.07,"output":0.33}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra (Sakana AI)","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"deepinfra/gemma-4-26b-a4b-it":{"id":"deepinfra/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT (DeepInfra)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"deepinfra/qwen3.5-9b":{"id":"deepinfra/qwen3.5-9b","name":"Qwen3.5 9B (DeepInfra)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.15}},"deepinfra/ling-3.0-flash":{"id":"deepinfra/ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash (DeepInfra)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"deepinfra/deepseek-v4-flash":{"id":"deepinfra/deepseek-v4-flash","name":"DeepSeek V4 Flash (DeepInfra)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.08,"output":0.18,"cache_read":0.016}},"deepinfra/hy3":{"id":"deepinfra/hy3","name":"Hy3 (DeepInfra)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":131072},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"deepinfra/deepseek-v3.2":{"id":"deepinfra/deepseek-v3.2","name":"DeepSeek V3.2 (DeepInfra)","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":65536},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"deepinfra/nemotron-3-ultra-550b":{"id":"deepinfra/nemotron-3-ultra-550b","name":"Nemotron 3 Ultra 550B (DeepInfra)","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"deepinfra/qwen3-vl-30b-a3b-instruct":{"id":"deepinfra/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct (DeepInfra)","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-10-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":0.6}},"deepinfra/gemma-4-31b-it":{"id":"deepinfra/gemma-4-31b-it","name":"Gemma 4 31B IT (DeepInfra)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"deepinfra/glm-5.1":{"id":"deepinfra/glm-5.1","name":"GLM-5.1 (DeepInfra)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":65536},"cost":{"input":1.05,"output":3.5,"cache_read":0.205}},"deepinfra/qwen3-vl-235b-a22b-instruct":{"id":"deepinfra/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct (DeepInfra)","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"deepinfra/deepseek-v4-pro":{"id":"deepinfra/deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepInfra)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepinfra/mimo-v2.5":{"id":"deepinfra/mimo-v2.5","name":"MiMo V2.5 (DeepInfra)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.4,"output":2,"cache_read":0.08,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"deepinfra/mimo-v2.5-pro":{"id":"deepinfra/mimo-v2.5-pro","name":"MiMo V2.5 Pro (DeepInfra)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"azure-ai-foundry/grok-4-1-fast-non-reasoning":{"id":"azure-ai-foundry/grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning (Azure AI Foundry)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"azure-ai-foundry/grok-4-1-fast-reasoning":{"id":"azure-ai-foundry/grok-4-1-fast-reasoning","name":"Grok 4.1 Fast Reasoning (Azure AI Foundry)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"azure-ai-foundry/grok-4-3":{"id":"azure-ai-foundry/grok-4-3","name":"Grok 4.3 (Azure AI Foundry)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":20000,"output":8192},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed (Moonshot AI)","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6 (Moonshot AI)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code (Moonshot AI)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3 (Moonshot AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshot/kimi-k2.5":{"id":"moonshot/kimi-k2.5","name":"Kimi K2.5 (Moonshot AI)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"consensusprotocol/Qwen3.8-27B":{"id":"consensusprotocol/Qwen3.8-27B","name":"Qwen3.8 27B (Consensus Protocol)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":2,"cache_read":0.05}},"consensusprotocol/deepseek-v4-flash":{"id":"consensusprotocol/deepseek-v4-flash","name":"DeepSeek V4 Flash (Consensus Protocol)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.05,"output":0.1,"cache_read":0.01}},"consensusprotocol/gpt-oss-20b":{"id":"consensusprotocol/gpt-oss-20b","name":"GPT OSS 20B (Consensus Protocol)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":0.04,"output":0.19,"cache_read":0.01}},"consensusprotocol/glm-5.3-flash":{"id":"consensusprotocol/glm-5.3-flash","name":"GLM-5.3 Flash (Consensus Protocol)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.25,"cache_read":0.02}},"consensusprotocol/gemma-4-31b-it":{"id":"consensusprotocol/gemma-4-31b-it","name":"Gemma 4 31B IT (Consensus Protocol)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.25,"cache_read":0.01}},"azure/gpt-5-nano":{"id":"azure/gpt-5-nano","name":"GPT-5 Nano (Azure)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"azure/gpt-4.1-nano":{"id":"azure/gpt-4.1-nano","name":"GPT-4.1 Nano (Azure)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"azure/gpt-5.1-codex-mini":{"id":"azure/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-5.1-codex":{"id":"azure/gpt-5.1-codex","name":"GPT-5.1 Codex (Azure)","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":1.25,"output":10}},"azure/gpt-5.6-sol":{"id":"azure/gpt-5.6-sol","name":"GPT-5.6 Sol (Azure)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"azure/gpt-5.2-codex":{"id":"azure/gpt-5.2-codex","name":"GPT-5.2 Codex (Azure)","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-6-astra":{"id":"azure/gpt-6-astra","name":"GPT-6 Astra (Azure)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"azure/gpt-5.2-pro":{"id":"azure/gpt-5.2-pro","name":"GPT-5.2 Pro (Azure)","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":21,"output":168}},"azure/gpt-4.1-mini":{"id":"azure/gpt-4.1-mini","name":"GPT-4.1 Mini (Azure)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"azure/gpt-5.4":{"id":"azure/gpt-5.4","name":"GPT-5.4 (Azure)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"azure/gpt-4-turbo":{"id":"azure/gpt-4-turbo","name":"GPT-4 Turbo (Azure)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"azure/gpt-5.1":{"id":"azure/gpt-5.1","name":"GPT-5.1 (Azure)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/o1":{"id":"azure/o1","name":"o1 (Azure)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"azure/gpt-4o":{"id":"azure/gpt-4o","name":"GPT-4o (Azure)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"azure/gpt-5.6-luna":{"id":"azure/gpt-5.6-luna","name":"GPT-5.6 Luna (Azure)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"azure/gpt-5.3-codex":{"id":"azure/gpt-5.3-codex","name":"GPT-5.3 Codex (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-4.1":{"id":"azure/gpt-4.1","name":"GPT-4.1 (Azure)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"azure/gpt-5.4-nano":{"id":"azure/gpt-5.4-nano","name":"GPT-5.4 Nano (Azure)","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"azure/gpt-5.4-mini":{"id":"azure/gpt-5.4-mini","name":"GPT-5.4 Mini (Azure)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"azure/gpt-3.5-turbo":{"id":"azure/gpt-3.5-turbo","name":"GPT-3.5 Turbo (Azure)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"azure/gpt-5-mini":{"id":"azure/gpt-5-mini","name":"GPT-5 Mini (Azure)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-oss-120b":{"id":"azure/gpt-oss-120b","name":"GPT OSS 120B (Azure)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"azure/gpt-5.4-pro":{"id":"azure/gpt-5.4-pro","name":"GPT-5.4 Pro (Azure)","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"azure/gpt-5.6-terra":{"id":"azure/gpt-5.6-terra","name":"GPT-5.6 Terra (Azure)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"azure/gpt-4":{"id":"azure/gpt-4","name":"GPT-4 (Azure)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"azure/gpt-5.2":{"id":"azure/gpt-5.2","name":"GPT-5.2 (Azure)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5":{"id":"azure/gpt-5","name":"GPT-5 (Azure)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/o4-mini":{"id":"azure/o4-mini","name":"o4 Mini (Azure)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"azure/o3-mini":{"id":"azure/o3-mini","name":"o3 Mini (Azure)","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"azure/o3":{"id":"azure/o3","name":"o3 (Azure)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"azure/gpt-5.5":{"id":"azure/gpt-5.5","name":"GPT-5.5 (Azure)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash (DeepSeek)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepSeek)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano (OpenAI)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 Nano (OpenAI)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro (OpenAI)","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-transcribe":{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe (OpenAI)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":2000},"cost":{"input":1.25,"output":5}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol (OpenAI)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra (OpenAI)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-03","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro (OpenAI)","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 Mini (OpenAI)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4 (OpenAI)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-4o-transcribe":{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe (OpenAI)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":2000},"cost":{"input":2.5,"output":10}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo (OpenAI)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1 (OpenAI)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1 (OpenAI)","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o (OpenAI)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna (OpenAI)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex (OpenAI)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o Mini (OpenAI)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1 (OpenAI)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano (OpenAI)","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro (OpenAI)","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini (OpenAI)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo (OpenAI)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini (OpenAI)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro (OpenAI)","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra (OpenAI)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4 (OpenAI)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2 (OpenAI)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5 (OpenAI)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4 Mini (OpenAI)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3 Mini (OpenAI)","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3 (OpenAI)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5 (OpenAI)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"meta-contributor/muse-spark-1.2-contributor":{"id":"meta-contributor/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor (Meta Contributor)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta-contributor/muse-spark-1.3-contributor":{"id":"meta-contributor/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor (Meta Contributor)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"xai/grok-4":{"id":"xai/grok-4","name":"Grok 4 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4-5":{"id":"xai/grok-4-5","name":"Grok 4.5 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-build-0-1":{"id":"xai/grok-build-0-1","name":"Grok Build 0.1 (xAI)","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4-3":{"id":"xai/grok-4-3","name":"Grok 4.3 (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-20-beta-0309-reasoning":{"id":"xai/grok-4-20-beta-0309-reasoning","name":"Grok 4.20 Beta Reasoning (0309) (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4-6":{"id":"xai/grok-4-6","name":"Grok 4.6 (xAI)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4-20-beta-0309-non-reasoning":{"id":"xai/grok-4-20-beta-0309-non-reasoning","name":"Grok 4.20 Beta Non-Reasoning (0309) (xAI)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7 (Z AI)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5 Air (Z AI)","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6 (Z AI)","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6v":{"id":"zai/glm-4.6v","name":"GLM-4.6V (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"zai/glm-4.6v-flashx":{"id":"zai/glm-4.6v-flashx","name":"GLM-4.6V FlashX (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.04,"output":0.4,"cache_read":0.004}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2 (Z AI)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-4.5-x":{"id":"zai/glm-4.5-x","name":"GLM-4.5 X (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":2.2,"output":8.9,"cache_read":0.45}},"zai/glm-4.5-airx":{"id":"zai/glm-4.5-airx","name":"GLM-4.5 AirX (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":1.1,"output":4.5,"cache_read":0.22}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3 Flash (Z AI)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5 (Z AI)","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"GLM-4.5V (Z AI)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM-4.7 FlashX (Z AI)","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.01}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5 (Z AI)","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131100},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-4-32b-0414-128k":{"id":"zai/glm-4-32b-0414-128k","name":"GLM-4 32B (0414-128k) (Z AI)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1 (Z AI)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3 (Z AI)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"azure-anthropic/claude-opus-5":{"id":"azure-anthropic/claude-opus-5","name":"Claude Opus 5 (Azure Anthropic)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-opus-4-6":{"id":"azure-anthropic/claude-opus-4-6","name":"Claude Opus 4.6 (Azure Anthropic)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-opus-4-7":{"id":"azure-anthropic/claude-opus-4-7","name":"Claude Opus 4.7 (Azure Anthropic)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-fable-5":{"id":"azure-anthropic/claude-fable-5","name":"Claude Fable 5 (Azure Anthropic)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"azure-anthropic/claude-opus-4-8":{"id":"azure-anthropic/claude-opus-4-8","name":"Claude Opus 4.8 (Azure Anthropic)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"azure-anthropic/claude-sonnet-5":{"id":"azure-anthropic/claude-sonnet-5","name":"Claude Sonnet 5 (Azure Anthropic)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"fireworks/deepseek-v4-flash":{"id":"fireworks/deepseek-v4-flash","name":"DeepSeek V4 Flash (Fireworks AI)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks/kimi-k3":{"id":"fireworks/kimi-k3","name":"Kimi K3 (Fireworks AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":1040384},"cost":{"input":3,"output":15,"cache_read":0.3}},"fireworks/kimi-k3-fast":{"id":"fireworks/kimi-k3-fast","name":"Kimi K3 Fast (Fireworks AI)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":1040384},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"fireworks/deepseek-v4-pro":{"id":"fireworks/deepseek-v4-pro","name":"DeepSeek V4 Pro (Fireworks AI)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"mistral/ministral-14b-2512":{"id":"mistral/ministral-14b-2512","name":"Ministral 14B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.2}},"mistral/codestral-2508":{"id":"mistral/codestral-2508","name":"Codestral (Mistral AI)","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"mistral/mistral-small-2506":{"id":"mistral/mistral-small-2506","name":"Mistral Small 3.2 (Mistral AI)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2 (Mistral AI)","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3 (Mistral AI)","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/ministral-3b-2512":{"id":"mistral/ministral-3b-2512","name":"Ministral 3B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large Latest (Mistral AI)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":4,"output":12}},"mistral/ministral-8b-2512":{"id":"mistral/ministral-8b-2512","name":"Ministral 8B (Mistral AI)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.15,"output":0.15}},"cerebras/glm-4.7":{"id":"cerebras/glm-4.7","name":"GLM-4.7 (Cerebras)","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":2.25,"output":2.75}},"cerebras/gemma-4-31b-it":{"id":"cerebras/gemma-4-31b-it","name":"Gemma 4 31B IT (Cerebras)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.99,"output":1.49}},"cerebras/qwen3-235b-a22b-instruct-2507":{"id":"cerebras/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507 (Cerebras)","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":8192},"cost":{"input":0.6,"output":1.2}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B (Cerebras)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75}},"cerebras/llama-3.3-70b-instruct":{"id":"cerebras/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct (Cerebras)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.85,"output":1.2}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar (Perplexity)","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":4096},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro (Perplexity)","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro (Perplexity)","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"runware/kimi-k2.6":{"id":"runware/kimi-k2.6","name":"Kimi K2.6 (Runware)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.6,"output":3.05,"cache_read":0.13}},"runware/glm-5.2":{"id":"runware/glm-5.2","name":"GLM-5.2 (Runware)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"runware/deepseek-v4-flash":{"id":"runware/deepseek-v4-flash","name":"DeepSeek V4 Flash (Runware)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.076,"output":0.153,"cache_read":0.014}},"runware/kimi-k3":{"id":"runware/kimi-k3","name":"Kimi K3 (Runware)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"runware/glm-5.3-flash":{"id":"runware/glm-5.3-flash","name":"GLM-5.3 Flash (Runware)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"runware/gemma-4-31b-it":{"id":"runware/gemma-4-31b-it","name":"Gemma 4 31B IT (Runware)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.102,"output":0.297,"cache_read":0.012}},"runware/deepseek-v4-pro":{"id":"runware/deepseek-v4-pro","name":"DeepSeek V4 Pro (Runware)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.961,"output":1.922,"cache_read":0.079}},"runware/gpt-oss-120b":{"id":"runware/gpt-oss-120b","name":"GPT OSS 120B (Runware)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.032,"output":0.14,"cache_read":0.032}},"runware/glm-5.3":{"id":"runware/glm-5.3","name":"GLM-5.3 (Runware)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}}}},"llama":{"id":"llama","env":["LLAMA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llama.com/compat/v1/","name":"Llama","doc":"https://llama.developer.meta.com/docs/models","models":{"cerebras-llama-4-scout-17b-16e-instruct":{"id":"cerebras-llama-4-scout-17b-16e-instruct","name":"Cerebras-Llama-4-Scout-17B-16E-Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama-4-Maverick-17B-128E-Instruct-FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"groq-llama-4-maverick-17b-128e-instruct":{"id":"groq-llama-4-maverick-17b-128e-instruct","name":"Groq-Llama-4-Maverick-17B-128E-Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-4-scout-17b-16e-instruct-fp8":{"id":"llama-4-scout-17b-16e-instruct-fp8","name":"Llama-4-Scout-17B-16E-Instruct-FP8","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"cerebras-llama-4-maverick-17b-128e-instruct":{"id":"cerebras-llama-4-maverick-17b-128e-instruct","name":"Cerebras-Llama-4-Maverick-17B-128E-Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-8b-instruct":{"id":"llama-3.3-8b-instruct","name":"Llama-3.3-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}}}},"alibaba-token-plan":{"id":"alibaba-token-plan","env":["ALIBABA_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1","name":"Alibaba Token Plan","doc":"https://www.alibabacloud.com/help/en/model-studio/token-plan-overview","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-r2v":{"id":"happyhorse-1.1-r2v","name":"HappyHorse 1.1 Reference-to-Video","description":"Video model for reference-guided video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max-preview":{"id":"qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image-pro":{"id":"wan2.7-image-pro","name":"Wan2.7 Image Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-t2v":{"id":"happyhorse-1.1-t2v","name":"HappyHorse 1.1 Text-to-Video","description":"Video model for prompt-driven text-to-video generation","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen-image-2.0-pro":{"id":"qwen-image-2.0-pro","name":"Qwen Image 2.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-03","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-i2v":{"id":"happyhorse-1.1-i2v","name":"HappyHorse 1.1 Image-to-Video","description":"Video model for image-to-video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen-image-2.0":{"id":"qwen-image-2.0","name":"Qwen Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image":{"id":"wan2.7-image","name":"Wan2.7 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}}}},"neuralwatt":{"id":"neuralwatt","env":["NEURALWATT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.neuralwatt.com/v1","name":"Neuralwatt","doc":"https://portal.neuralwatt.com/docs","models":{"glm-5.2-short-fast-flex":{"id":"glm-5.2-short-fast-flex","name":"GLM 5.2 Short Fast Flex","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"glm-5.2-short-flex":{"id":"glm-5.2-short-flex","name":"GLM 5.2 Short Flex","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"glm-5.2-flex":{"id":"glm-5.2-flex","name":"GLM 5.2 Flex","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":0.9425,"output":2.925,"cache_read":0.09425}},"glm-5.2":{"id":"glm-5.2","name":"GLM 5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":65536},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.95,"output":4,"cache_read":0.095}},"kimi-k2.7-code-flex":{"id":"kimi-k2.7-code-flex","name":"Kimi K2.7 Code Flex","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.6175,"output":2.6,"cache_read":0.06175}},"kimi-k2.7-code-fast":{"id":"kimi-k2.7-code-fast","name":"Kimi K2.7 Code Fast","description":"Kimi K2.7 Code with reasoning capped to a short budget for lower latency; reasoning cannot be disabled on this model","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":262128},"cost":{"input":0.95,"output":4,"cache_read":0.095}},"glm-5.2-short-fast":{"id":"glm-5.2-short-fast","name":"GLM 5.2 Short Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k3-fast":{"id":"kimi-k3-fast","name":"Kimi K3 Fast","description":"Kimi K3 with thinking disabled for low-latency tool calling, vision, and JSON work","family":"kimi-k3","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":3,"output":15,"cache_read":0.3}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"kimi-k3-flex":{"id":"kimi-k3-flex","name":"Kimi K3 Flex","description":"Kimi K3 on the flex tier: discounted, best-effort latency, requests may be held under load","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"cost":{"input":1.95,"output":9.75,"cache_read":0.195}},"qwen3.6-35b-fast":{"id":"qwen3.6-35b-fast","name":"Qwen3.6 35B Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen3.6","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.29,"output":1.15,"cache_read":0.029}},"gemma-4-31b":{"id":"gemma-4-31b","name":"Gemma 4 31B","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":16384},"cost":{"input":0.144,"output":0.42,"cache_read":0.0144}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":393216},"status":"beta","cost":{"input":1,"output":3,"cache_read":0.1}},"deepseek-v4-flash-flex":{"id":"deepseek-v4-flash-flex","name":"DeepSeek V4 Flash Flex","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":65536},"cost":{"input":0.091,"output":0.182,"cache_read":0.0182}},"glm-5.3":{"id":"glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048560,"output":1048560},"status":"beta","cost":{"input":1.45,"output":4.5,"cache_read":0.145}},"qwen3.6-35b":{"id":"qwen3.6-35b","name":"Qwen3.6 35B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131056,"output":131056},"cost":{"input":0.29,"output":1.15,"cache_read":0.029}},"qwen-3.8-27b":{"id":"qwen-3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262128,"output":65536},"status":"beta","cost":{"input":0.45,"output":3.2,"cache_read":0.25}},"glm-5.2-short":{"id":"glm-5.2-short","name":"GLM 5.2 Short","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":199984,"output":32000},"cost":{"input":1.45,"output":4.5,"cache_read":0.145}}}},"abliteration-ai":{"id":"abliteration-ai","env":["ABLIT_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.abliteration.ai/v1","name":"abliteration.ai","doc":"https://docs.abliteration.ai/models","models":{"abliterated-model-large":{"id":"abliterated-model-large","name":"Abliterated Model Large","description":"GLM-5.2 model abliterated and finetuned for cyber, ML red teaming, and agent testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]},{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-25","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliterated-model-large-v2":{"id":"abliterated-model-large-v2","name":"Abliterated Model Large V2","description":"GLM-5.3 model abliterated and finetuned for cyber, ML red teaming, and agent testing","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":1000000,"output":999990},"cost":{"input":5,"output":5,"cache_read":0.5}},"abliterated-model":{"id":"abliterated-model","name":"Abliterated Model","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]},{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01-06","last_updated":"2026-07-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":150000,"input":150000,"output":8192},"cost":{"input":3,"output":3,"cache_read":0.3}}}},"clarifai":{"id":"clarifai","env":["CLARIFAI_PAT"],"npm":"@ai-sdk/openai-compatible","api":"https://api.clarifai.com/v2/ext/openai/v1","name":"Clarifai","doc":"https://docs.clarifai.com/compute/inference/","models":{"qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct":{"id":"qwen/qwenCoder/models/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.11458,"output":0.74812}},"qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507":{"id":"qwen/qwenLM/models/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.5}},"qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507":{"id":"qwen/qwenLM/models/Qwen3-30B-A3B-Thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.36,"output":1.3}},"clarifai/main/models/mm-poly-8b":{"id":"clarifai/main/models/mm-poly-8b","name":"MM Poly 8B","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"mm-poly","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.658,"output":1.11}},"deepseek-ai/deepseek-ocr/models/DeepSeek-OCR":{"id":"deepseek-ai/deepseek-ocr/models/DeepSeek-OCR","name":"DeepSeek OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"deepseek","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2026-02-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.2,"output":0.7}},"mistralai/completion/models/Ministral-3-14B-Reasoning-2512":{"id":"mistralai/completion/models/Ministral-3-14B-Reasoning-2512","name":"Ministral 3 14B Reasoning 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-01","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2.5,"output":1.7}},"mistralai/completion/models/Ministral-3-3B-Reasoning-2512":{"id":"mistralai/completion/models/Ministral-3-3B-Reasoning-2512","name":"Ministral 3 3B Reasoning 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12","last_updated":"2026-02-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.039,"output":0.54825}},"minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput":{"id":"minimaxai/chat-completion/models/MiniMax-M2_5-high-throughput","name":"MiniMax-M2.5 High Throughput","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"openai/chat-completion/models/gpt-oss-120b-high-throughput":{"id":"openai/chat-completion/models/gpt-oss-120b-high-throughput","name":"GPT OSS 120B High Throughput","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.36}},"openai/chat-completion/models/gpt-oss-20b":{"id":"openai/chat-completion/models/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-12-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.045,"output":0.18}},"moonshotai/chat-completion/models/Kimi-K2_6":{"id":"moonshotai/chat-completion/models/Kimi-K2_6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"arcee_ai/AFM/models/trinity-mini":{"id":"arcee_ai/AFM/models/trinity-mini","name":"Trinity Mini","description":"Reasoning-tuned 26B MoE model with 3B active parameters for agents, tools, and multi-step workloads","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-01","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.045,"output":0.15}}}},"morph":{"id":"morph","env":["MORPH_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.morphllm.com/v1","name":"Morph","doc":"https://docs.morphllm.com/api-reference/introduction","models":{"morph-v3-large":{"id":"morph-v3-large","name":"Morph v3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.9,"output":1.9}},"morph-v3-fast":{"id":"morph-v3-fast","name":"Morph v3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":0.8,"output":1.2}},"auto":{"id":"auto","name":"Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.85,"output":1.55}}}},"aihubmix":{"id":"aihubmix","env":["AIHUBMIX_API_KEY"],"npm":"@aihubmix/ai-sdk-provider","name":"AIHubMix","doc":"https://docs.aihubmix.com","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":1.69,"output":5.07,"cache_read":0.169,"cache_write":2.1125}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":7.999,"cache_read":0.32167}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.6918,"output":2.0754,"cache_read":0.023058}},"doubao-seed-2-0-lite-260428":{"id":"doubao-seed-2-0-lite-260428","name":"Doubao Seed 2.0 Lite 260428","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.08,"output":0.51,"cache_read":0.01692,"input_audio":1.269,"tiers":[{"input":0.13,"output":0.76,"cache_read":0.02536,"input_audio":1.902,"tier":{"type":"context","size":32000}},{"input":0.25,"output":1.52,"cache_read":0.05072,"input_audio":3.804,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.142,"output":0.284,"cache_read":0.0284}},"coding-minimax-m2.7":{"id":"coding-minimax-m2.7","name":"Coding MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0.2,"output":0.2}},"coding-glm-5.1":{"id":"coding-glm-5.1","name":"Coding GLM 5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-11","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.06,"output":0.22,"cache_read":0.013}},"claude-opus-4-7-think":{"id":"claude-opus-4-7-think","name":"Claude Opus 4.7 Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.28,"output":1.69,"cache_read":0.0282,"cache_write":0.3525,"tiers":[{"input":1.13,"output":6.77,"cache_read":0.1128,"cache_write":1.41,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.13,"output":6.77,"cache_read":0.1128,"cache_write":1.41}}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"doubao-seed-2-0-mini-260428":{"id":"doubao-seed-2-0-mini-260428","name":"Doubao Seed 2.0 Mini 260428","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.03,"output":0.28,"cache_read":0.00564,"input_audio":0.423,"tiers":[{"input":0.06,"output":0.56,"cache_read":0.01128,"input_audio":0.846,"tier":{"type":"context","size":32000}},{"input":0.11,"output":1.13,"cache_read":0.02256,"input_audio":1.692,"tier":{"type":"context","size":128000}}]}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"doubao-seed-2-0-code-preview":{"id":"doubao-seed-2-0-code-preview","name":"Doubao Seed 2.0 Code Preview","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.48,"output":2.41,"cache_read":0.09644,"tiers":[{"input":0.72,"output":3.62,"cache_read":0.144656,"tier":{"type":"context","size":32000}},{"input":1.45,"output":7.23,"cache_read":0.28932,"tier":{"type":"context","size":128000}}]}},"xiaomi-mimo-v2.5-free":{"id":"xiaomi-mimo-v2.5-free","name":"Xiaomi MiMo-V2.5 (free)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"coding-xiaomi-mimo-v2.5-pro":{"id":"coding-xiaomi-mimo-v2.5-pro","name":"Coding Xiaomi MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.2,"output":0.6,"cache_read":0.04,"tiers":[{"input":0.4,"output":1.2,"cache_read":0.08,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.4,"output":1.2,"cache_read":0.08}}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1268,"output":3.9438,"cache_read":0.2817}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":3.9995,"cache_read":0.160835}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.5}},"alicloud-deepseek-v4-pro":{"id":"alicloud-deepseek-v4-pro","name":"DeepSeek V4 Pro (Alibaba Cloud)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.69,"output":3.38,"cache_read":0.13}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xiaomi-mimo-v2.5-pro-free":{"id":"xiaomi-mimo-v2.5-pro-free","name":"Xiaomi MiMo-V2.5-Pro (free)","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"deep-deepseek-v4-pro":{"id":"deep-deepseek-v4-pro","name":"DeepSeek V4 Pro (DeepSeek)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.478,"output":0.956,"cache_read":0.004302}},"deep-deepseek-v4-flash":{"id":"deep-deepseek-v4-flash","name":"DeepSeek V4 Flash (DeepSeek)","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.154,"output":0.308,"cache_read":0.0308}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":1.5}},"xiaomi-mimo-v2.5":{"id":"xiaomi-mimo-v2.5","name":"Xiaomi MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.088,"tiers":[{"input":0.88,"output":4.4,"cache_read":0.176,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.88,"output":4.4,"cache_read":0.176}}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"input":991000,"output":64000},"cost":{"input":0.0282,"output":0.1128,"cache_read":0.00564,"cache_write":0.03525}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"alicloud-deepseek-v4-flash":{"id":"alicloud-deepseek-v4-flash","name":"DeepSeek V4 Flash (Alibaba Cloud)","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"zai-glm-5.1":{"id":"zai-glm-5.1","name":"GLM-5.1 (Z.ai)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.845,"output":3.38,"cache_read":0.183112}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.11268,"output":0.39438,"cache_read":0.02817}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":2,"output":6,"cache_read":0.5}},"claude-opus-4-8-think":{"id":"claude-opus-4-8-think","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"coding-xiaomi-mimo-v2.5":{"id":"coding-xiaomi-mimo-v2.5","name":"Coding Xiaomi MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.08,"output":0.4,"cache_read":0.016,"tiers":[{"input":0.16,"output":0.8,"cache_read":0.032,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.16,"output":0.8,"cache_read":0.032}}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.17,"output":1.01,"cache_read":0.0169,"cache_write":0.21125,"tiers":[{"input":0.68,"output":4.06,"cache_read":0.0676,"cache_write":0.845,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.68,"output":4.06,"cache_read":0.0676,"cache_write":0.845}}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"coding-minimax-m2.7-free":{"id":"coding-minimax-m2.7-free","name":"Coding MiniMax M2.7 (Free)","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0,"output":0}},"doubao-seed-2-0-pro":{"id":"doubao-seed-2-0-pro","name":"Doubao Seed 2.0 Pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.48,"output":2.41,"cache_read":0.09644,"tiers":[{"input":0.72,"output":3.62,"cache_read":0.144656,"tier":{"type":"context","size":32000}},{"input":1.45,"output":7.23,"cache_read":0.28932,"tier":{"type":"context","size":128000}}]}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":240000,"output":64000},"cost":{"input":1.27,"output":7.61,"cache_read":0.1268,"cache_write":1.585,"tiers":[{"input":2.11,"output":12.67,"cache_read":0.2112,"cache_write":2.64,"tier":{"type":"context","size":128000}}]}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"alicloud-glm-5.1":{"id":"alicloud-glm-5.1","name":"GLM-5.1 (Alibaba Cloud)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.84,"output":3.38,"cache_read":0.169,"cache_write":1.05625}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":1.69,"output":5.07,"cache_read":0.169,"cache_write":2.1125}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05}}},"claude-sonnet-4-6-think":{"id":"claude-sonnet-4-6-think","name":"Claude Sonnet 4.6 Thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.282,"output":1.128,"cache_read":0.0564,"cache_write":0.3525}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"hy3-preview":{"id":"hy3-preview","name":"Hy3 Preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.17,"output":0.566661,"cache_read":0.051}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1268,"output":3.9438,"cache_read":0.2817}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM 5 Vision Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-09","last_updated":"2026-05-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.7042,"output":3.09848,"cache_read":0.169008}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"xiaomi-mimo-v2.5-pro":{"id":"xiaomi-mimo-v2.5-pro","name":"Xiaomi MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo-v2.5-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.1,"output":3.3,"cache_read":0.22,"tiers":[{"input":2.2,"output":6.6,"cache_read":0.44,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2.2,"output":6.6,"cache_read":0.44}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-6-think":{"id":"claude-opus-4-6-think","name":"Claude Opus 4.6 Thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"coding-glm-5.1-free":{"id":"coding-glm-5.1-free","name":"Coding GLM 5.1 (free)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-11","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"coding-minimax-m2.7-highspeed":{"id":"coding-minimax-m2.7-highspeed","name":"Coding MiniMax M2.7 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128100},"cost":{"input":0.2,"output":0.2}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"chutes":{"id":"chutes","env":["CHUTES_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llm.chutes.ai/v1","name":"Chutes","doc":"https://llm.chutes.ai/v1/models","models":{"Nemotron-3-Nano-Omni-30B-TEE":{"id":"Nemotron-3-Nano-Omni-30B-TEE","name":"Nemotron 3 Nano Omni 30B TEE","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":0},"cost":{"input":0.0245,"output":0.0978,"cache_read":0.0024499999999999995}},"deepseek-ai/DeepSeek-V4-Flash-0731-TEE":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731-TEE","name":"DeepSeek V4 Flash 0731 TEE","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.44,"output":1.32,"cache_read":0.04399999999999999}},"deepseek-ai/DeepSeek-V3.2-TEE":{"id":"deepseek-ai/DeepSeek-V3.2-TEE","name":"DeepSeek V3.2 TEE","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12","last_updated":"2026-06-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":1,"output":1,"cache_read":0.09999999999999998}},"google/gemma-4-31B-turbo-TEE":{"id":"google/gemma-4-31B-turbo-TEE","name":"gemma 4 31B turbo TEE","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.12,"output":0.37,"cache_read":0.011999999999999997}},"zai-org/GLM-5.1-TEE":{"id":"zai-org/GLM-5.1-TEE","name":"GLM 5.1 TEE","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":65535},"cost":{"input":0.98,"output":3.08,"cache_read":0.09799999999999998}},"zai-org/GLM-5.2-TEE":{"id":"zai-org/GLM-5.2-TEE","name":"GLM 5.2 TEE","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":3.95,"cache_read":0.12499999999999997}},"Qwen/Qwen3.8-27B-TEE":{"id":"Qwen/Qwen3.8-27B-TEE","name":"Qwen3.8 27B TEE","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-16","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":2.5,"cache_read":0.031999999999999994}},"Qwen/Qwen3.6-27B-TEE":{"id":"Qwen/Qwen3.6-27B-TEE","name":"Qwen3.6 27B TEE","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2,"cache_read":0.029999999999999992}},"Qwen/Qwen3.5-397B-A17B-TEE":{"id":"Qwen/Qwen3.5-397B-A17B-TEE","name":"Qwen3.5 397B A17B TEE","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3,"cache_read":0.04499999999999999}},"Qwen/Qwen3-235B-A22B-Thinking-2507-TEE":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507-TEE","name":"Qwen3 235B A22B Thinking 2507 TEE","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07","last_updated":"2026-06-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2989,"output":1.1957,"cache_read":0.029889999999999993}},"Qwen/Qwen3-32B-TEE":{"id":"Qwen/Qwen3-32B-TEE","name":"Qwen3 32B TEE","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.104,"output":0.416,"cache_read":0.010399999999999998}},"unsloth/Mistral-Nemo-Instruct-2407-TEE":{"id":"unsloth/Mistral-Nemo-Instruct-2407-TEE","name":"Mistral Nemo Instruct 2407 TEE","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0245,"output":0.0978,"cache_read":0.0024499999999999995}},"moonshotai/Kimi-K3-TEE":{"id":"moonshotai/Kimi-K3-TEE","name":"Kimi K3 TEE","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65535},"cost":{"input":3,"output":15,"cache_read":0.29999999999999993}},"moonshotai/Kimi-K2.6-TEE":{"id":"moonshotai/Kimi-K2.6-TEE","name":"Kimi K2.6 TEE","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65535},"cost":{"input":0.58,"output":3.4,"cache_read":0.05799999999999998}}}},"groq":{"id":"groq","env":["GROQ_API_KEY"],"npm":"@ai-sdk/groq","name":"Groq","doc":"https://console.groq.com/docs/models","models":{"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-09-01","last_updated":"2025-09-05","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":0}},"whisper-large-v3-turbo":{"id":"whisper-large-v3-turbo","name":"Whisper Large V3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":0}},"llama-3.1-8b-instant":{"id":"llama-3.1-8b-instant","name":"Llama 3.1 8B","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.08}},"allam-2-7b":{"id":"allam-2-7b","name":"ALLaM-2-7b","description":"ALLaM-2-7b instruction tuned model by SDAIA","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0,"output":0}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.59,"output":0.79}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","default","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131042,"output":16384},"cost":{"input":0.8,"output":4}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","default"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.6,"output":3,"cache_read":0.3}},"groq/compound":{"id":"groq/compound","name":"Compound","description":"General-purpose chat model for instruction following, writing, and analysis","family":"groq","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192}},"groq/compound-mini":{"id":"groq/compound-mini","name":"Compound Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"groq","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192}},"meta-llama/llama-prompt-guard-2-86m":{"id":"meta-llama/llama-prompt-guard-2-86m","name":"Prompt Guard 2 86M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"status":"beta","cost":{"input":0.04,"output":0.04}},"meta-llama/llama-prompt-guard-2-22m":{"id":"meta-llama/llama-prompt-guard-2-22m","name":"Llama Prompt Guard 2 22M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":512},"status":"beta","cost":{"input":0.03,"output":0.03}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"Safety GPT OSS 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"status":"beta","cost":{"input":0.075,"output":0.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-10-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"canopylabs/orpheus-v1-english":{"id":"canopylabs/orpheus-v1-english","name":"Canopy Labs Orpheus V1 English","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"canopylabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":4000,"output":50000},"status":"beta"},"canopylabs/orpheus-arabic-saudi":{"id":"canopylabs/orpheus-arabic-saudi","name":"Canopy Labs Orpheus Arabic Saudi","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"canopylabs","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":4000,"output":50000},"status":"beta"}}},"zai-coding-plan":{"id":"zai-coding-plan","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.z.ai/api/coding/paas/v4","name":"Z.AI Coding Plan","doc":"https://docs.z.ai/devpack/overview","models":{"glm-5.2-highspeed":{"id":"glm-5.2-highspeed","name":"GLM-5.2 Highspeed","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-highspeed":{"id":"glm-5.3-highspeed","name":"GLM-5.3 Highspeed","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"volcengine":{"id":"volcengine","env":["ARK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ark.cn-beijing.volces.com/api/v3","name":"Volcengine Ark","doc":"https://www.volcengine.com/docs/82379/1330310","models":{"doubao-seed-2-0-lite-260428":{"id":"doubao-seed-2-0-lite-260428","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.08906,"output":0.53436,"cache_read":0.01781,"tiers":[{"input":0.13359,"output":0.80154,"cache_read":0.02672,"tier":{"type":"context","size":32000}},{"input":0.26718,"output":1.60308,"cache_read":0.05344,"tier":{"type":"context","size":128000}}]}},"doubao-seed-character-260628":{"id":"doubao-seed-character-260628","name":"Seed Character","description":"ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.11875,"output":0.29687,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":0.8906,"cache_read":0.02375,"tier":{"type":"context","size":32000}}]}},"doubao-seed-2-0-mini-260428":{"id":"doubao-seed-2-0-mini-260428","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.02969,"output":0.29687,"cache_read":0.00594,"tiers":[{"input":0.05937,"output":0.59374,"cache_read":0.01187,"tier":{"type":"context","size":32000}},{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-2-1-pro-260628":{"id":"doubao-seed-2-1-pro-260628","name":"Seed 2.1 Pro","description":"Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.8906,"output":4.45301,"cache_read":0.17812}},"doubao-seed-2-0-code-preview-260215":{"id":"doubao-seed-2-0-code-preview-260215","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.47499,"output":2.37494,"cache_read":0.095,"tiers":[{"input":0.71248,"output":3.56241,"cache_read":0.1425,"tier":{"type":"context","size":32000}},{"input":1.42496,"output":7.12482,"cache_read":0.28499,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-8-251228":{"id":"doubao-seed-1-8-251228","name":"Seed 1.8","description":"ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-28","last_updated":"2025-12-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-6-flash-250828":{"id":"doubao-seed-1-6-flash-250828","name":"Seed 1.6 Flash","description":"Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.02227,"output":0.22265,"cache_read":0.00445,"tiers":[{"input":0.04453,"output":0.4453,"cache_read":0.00445,"tier":{"type":"context","size":32000}},{"input":0.08906,"output":0.8906,"cache_read":0.00445,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-6-251015":{"id":"doubao-seed-1-6-251015","name":"Seed 1.6","description":"ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"deepseek-v4-pro-ga-260813":{"id":"deepseek-v4-pro-ga-260813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.3359,"output":4.00771,"cache_read":0.04453}},"doubao-seed-evolving":{"id":"doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.8906,"output":4.45301,"cache_read":0.17812}},"doubao-seed-2-0-pro-260215":{"id":"doubao-seed-2-0-pro-260215","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.47499,"output":2.37494,"cache_read":0.095,"tiers":[{"input":0.71248,"output":3.56241,"cache_read":0.1425,"tier":{"type":"context","size":32000}},{"input":1.42496,"output":7.12482,"cache_read":0.28499,"tier":{"type":"context","size":128000}}]}},"doubao-seed-1-6-vision-250815":{"id":"doubao-seed-1-6-vision-250815","name":"Seed 1.6 Vision","description":"ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.11875,"output":1.18747,"cache_read":0.02375,"tiers":[{"input":0.17812,"output":2.37494,"cache_read":0.02375,"tier":{"type":"context","size":32000}},{"input":0.35624,"output":3.56241,"cache_read":0.02375,"tier":{"type":"context","size":128000}}]}},"glm-5-2-260617":{"id":"glm-5-2-260617","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.18747,"output":4.15615,"cache_read":0.29687}},"doubao-seed-2-1-turbo-260628":{"id":"doubao-seed-2-1-turbo-260628","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.4453,"output":2.22651,"cache_read":0.08906}},"deepseek-v4-flash-ga-260731":{"id":"deepseek-v4-flash-ga-260731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.4453,"output":1.3359,"cache_read":0.01484}}}},"sensenova":{"id":"sensenova","env":["SENSENOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token.sensenova.cn/v1","name":"SenseNova (China)","doc":"https://platform.sensenova.cn/docs","models":{"sensenova-6.8-flash-lite":{"id":"sensenova-6.8-flash-lite","name":"SenseNova 6.8 Flash Lite","description":"SenseNova lightweight multimodal agent model for real-world complex tasks, data analysis, and complex information presentation","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0,"cache_read":0}}}},"orcarouter":{"id":"orcarouter","env":["ORCAROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.orcarouter.ai/v1","name":"OrcaRouter","doc":"https://docs.orcarouter.ai","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.563}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.086,"output":0.688}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.33,"output":2.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.057,"output":0.459}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.172,"output":1.032}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.115,"output":0.917}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":40960},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.35,"output":1.42,"cache_read":0.071}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.115,"output":0.688,"reasoning":2.4}},"orcarouter/free":{"id":"orcarouter/free","name":"OrcaRouter Free","description":"Built-in router over the free tier that scores each request's difficulty and sends light work to the smaller free model and harder work to the stronger one. Priced at zero and never falls back to a paid model.","family":"auto","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":0,"output":0}},"orcarouter/fusion-mini":{"id":"orcarouter/fusion-mini","name":"OrcaRouter Fusion Mini","description":"Leaner two-model Fusion panel that runs Claude Opus 4.8 and GPT-5.5 in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Easy requests fall through to a cheaper default and bill as one call.","family":"model-router","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"orcarouter/fusion":{"id":"orcarouter/fusion","name":"OrcaRouter Fusion","description":"Curated fan-out router that runs Claude Opus 4.8, GPT-5.5 and Gemini 3.1 Pro in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Easy requests fall through to a cheaper default and bill as one call.","family":"model-router","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"orcarouter/fusion-flash":{"id":"orcarouter/fusion-flash","name":"OrcaRouter Fusion Flash","description":"Budget Fusion panel that runs Gemini 3.5 Flash, MiniMax M2.7 and GLM 5.1 in parallel on hard requests, then has a Claude Opus 4.8 judge return the strongest single answer verbatim. Cost-sensitive fan-out over a 200K window.","family":"model-router","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000}},"orcarouter/auto":{"id":"orcarouter/auto","name":"OrcaRouter Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2026-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":10}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.33,"cache_read":0.0075}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333,"input_audio":3}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-robotics-er-1.6-preview":{"id":"google/gemini-robotics-er-1.6-preview","name":"Gemini Robotics-ER 1.6 Preview","description":"Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1,"output":5}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38,"cache_read":0.02}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"grok/grok-4.3":{"id":"grok/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok/grok-4.5":{"id":"grok/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"grok/grok-4.6":{"id":"grok/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.442,"output":0.884,"cache_read":0.06}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-flash-free":{"id":"deepseek/deepseek-v4-flash-free","name":"DeepSeek V4 Flash (free)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-reasoner":{"id":"deepseek/deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.028}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.147,"output":0.295,"cache_read":0.02}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.442,"output":0.884,"cache_read":0.06}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5-chat-latest":{"id":"openai/gpt-5-chat-latest","name":"GPT-5 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":100000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-chat-latest":{"id":"openai/gpt-5.1-chat-latest","name":"GPT-5.1 Chat","description":"Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":100000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.17}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2-chat-latest":{"id":"openai/gpt-5.2-chat-latest","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"kimi/kimi-k2.6":{"id":"kimi/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi/kimi-k2.7-code":{"id":"kimi/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi/kimi-k3":{"id":"kimi/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3.3,"output":16.5,"cache_read":0.33}},"kimi/kimi-k2.5":{"id":"kimi/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3,"cache_read":0.1,"cache_write":0}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.18,"output":0.59,"cache_read":0.059}},"tencent/hy3-free":{"id":"tencent/hy3-free","name":"Hy3 (free)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.075,"output":0.25}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.26,"cache_write":0}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.26,"output":3.96,"cache_read":0.234}},"z-ai/glm-5.3-flash-free":{"id":"z-ai/glm-5.3-flash-free","name":"GLM-5.3-Flash (free)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}}}},"routing-run":{"id":"routing-run","env":["ROUTING_RUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.routing.run/v1","name":"routing.run","doc":"https://docs.routing.run/api-reference/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.16,"output":0.48}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.8,"output":2.4}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.112,"output":0.224}},"kimi-k2.6-nitro":{"id":"kimi-k2.6-nitro","name":"Kimi K2.6 Nitro","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"nemotron-3-ultra":{"id":"nemotron-3-ultra","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32000},"cost":{"input":0.1,"output":0.1}},"glm-5.2-nitro":{"id":"glm-5.2-nitro","name":"GLM 5.2 Nitro","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.8,"output":2.4}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":0.7,"output":4.2}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":5,"output":25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":64000},"cost":{"input":0.348,"output":0.696}},"kimi-k2.7-code-nitro":{"id":"kimi-k2.7-code-nitro","name":"Kimi K2.7 Code Nitro","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.275,"output":1.1}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":1.5,"output":9}}}},"llmtech":{"id":"llmtech","env":["LLMTECH_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmtech.eu/v1","name":"LLM Tech","doc":"https://llmtech.eu/models/qwen3.8-27b","models":{"unsloth/Qwen3.8-27B-NVFP4":{"id":"unsloth/Qwen3.8-27B-NVFP4","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2.09,"cache_read":0.04}}}},"sap-ai-core":{"id":"sap-ai-core","env":["AICORE_SERVICE_KEY"],"npm":"@jerome-benoit/sap-ai-provider-v2","name":"SAP AI Core","doc":"https://help.sap.com/docs/sap-ai-core","models":{"gpt-5-nano":{"id":"gpt-5-nano","name":"gpt-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.08,"output":0.26}},"anthropic--claude-4.5-sonnet":{"id":"anthropic--claude-4.5-sonnet","name":"anthropic--claude-4.5-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistralai--mistral-medium":{"id":"mistralai--mistral-medium","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"gpt-5.6-sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"anthropic--claude-4.5-haiku":{"id":"anthropic--claude-4.5-haiku","name":"anthropic--claude-4.5-haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"gemini-2.5-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"cohere--command-a-reasoning":{"id":"cohere--command-a-reasoning","name":"cohere--command-a-reasoning","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high"]},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.63,"output":5.05}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-5.4":{"id":"gpt-5.4","name":"gpt-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"anthropic--claude-4.8-opus":{"id":"anthropic--claude-4.8-opus","name":"anthropic--claude-4.8-opus","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"gemini-3.1-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"gemini-3.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"gpt-5.6-luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"tiers":[{"input":2,"output":9,"cache_read":0.2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2}}},"amazon--titan-embed-text":{"id":"amazon--titan-embed-text","name":"amazon--titan-embed-text","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-04-30","last_updated":"2024-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.14,"output":0}},"anthropic--claude-4.5-opus":{"id":"anthropic--claude-4.5-opus","name":"anthropic--claude-4.5-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic--claude-3.5-sonnet":{"id":"anthropic--claude-3.5-sonnet","name":"anthropic--claude-3.5-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistralai--mistral-medium-instruct":{"id":"mistralai--mistral-medium-instruct","name":"mistralai--mistral-medium-instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.36,"output":1.22}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"gpt-4.1":{"id":"gpt-4.1","name":"gpt-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.32}},"anthropic--claude-4.6-opus":{"id":"anthropic--claude-4.6-opus","name":"anthropic--claude-4.6-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"sonar":{"id":"sonar","name":"sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"anthropic--claude-4-sonnet":{"id":"anthropic--claude-4-sonnet","name":"anthropic--claude-4-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistralai--mistral-small":{"id":"mistralai--mistral-small","name":"mistralai--mistral-small","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.07,"output":0.28}},"amazon--nova-pro":{"id":"amazon--nova-pro","name":"amazon--nova-pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.56,"output":2.13}},"anthropic--claude-3-opus":{"id":"anthropic--claude-3-opus","name":"anthropic--claude-3-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-02-29","last_updated":"2024-02-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"nvidia--llama-3.2-nv-embedqa-1b":{"id":"nvidia--llama-3.2-nv-embedqa-1b","name":"nvidia--llama-3.2-nv-embedqa-1b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.07,"output":0}},"anthropic--claude-4.7-opus":{"id":"anthropic--claude-4.7-opus","name":"anthropic--claude-4.7-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-embedding-2":{"id":"gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":3072}},"sap-abap-1":{"id":"sap-abap-1","name":"sap-abap-1","description":"SAP-hosted model for ABAP code generation and enterprise development tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-26","last_updated":"2025-11-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.48,"output":1.7}},"amazon--nova-lite":{"id":"amazon--nova-lite","name":"amazon--nova-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.3,"output":2.37}},"anthropic--claude-3-haiku":{"id":"anthropic--claude-3-haiku","name":"anthropic--claude-3-haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"sonar-pro":{"id":"sonar-pro","name":"sonar-pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"amazon--nova-micro":{"id":"amazon--nova-micro","name":"amazon--nova-micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.03,"output":0.1}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"sonar-deep-research":{"id":"sonar-deep-research","name":"sonar-deep-research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.09,"output":0}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-25","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"gpt-5.6-terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":9.44,"cache_read":0.12}},"anthropic--claude-4.6-sonnet":{"id":"anthropic--claude-4.6-sonnet","name":"anthropic--claude-4.6-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5":{"id":"gpt-5","name":"gpt-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-04-17","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"anthropic--claude-4-opus":{"id":"anthropic--claude-4-opus","name":"anthropic--claude-4-opus","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5.5":{"id":"gpt-5.5","name":"gpt-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"anthropic--claude-3-sonnet":{"id":"anthropic--claude-3-sonnet","name":"anthropic--claude-3-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-04","last_updated":"2024-03-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-embedding":{"id":"gemini-embedding","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1}},"anthropic--claude-3.7-sonnet":{"id":"anthropic--claude-3.7-sonnet","name":"anthropic--claude-3.7-sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}}}},"alibaba-coding-plan-cn":{"id":"alibaba-coding-plan-cn","env":["ALIBABA_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding.dashscope.aliyuncs.com/v1","name":"Alibaba Coding Plan (China)","doc":"https://help.aliyun.com/zh/model-studio/coding-plan","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":24576},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"azure-cognitive-services":{"id":"azure-cognitive-services","env":["AZURE_COGNITIVE_SERVICES_RESOURCE_NAME","AZURE_COGNITIVE_SERVICES_API_KEY"],"npm":"@ai-sdk/azure","name":"Azure Cognitive Services","doc":"https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-07-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25,"tiers":[{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-chat-latest":{"id":"gpt-chat-latest","name":"GPT Chat Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"status":"beta","cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.6,"output":3}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-mythos-5":{"id":"claude-mythos-5","name":"Claude Mythos 5","description":"Restricted Claude model for advanced cybersecurity and biology research workflows","family":"claude-mythos","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_COGNITIVE_SERVICES_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.71,"output":0.71}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.125}},"gpt-3.5-turbo-instruct":{"id":"gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-09-21","last_updated":"2023-09-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096},"status":"deprecated","cost":{"input":1.5,"output":2}},"codestral-2501":{"id":"codestral-2501","name":"Codestral 25.01","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2}},"phi-4-reasoning":{"id":"phi-4-reasoning","name":"Phi-4-reasoning","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"deepseek-v3.2-speciale":{"id":"deepseek-v3.2-speciale","name":"DeepSeek-V3.2-Speciale","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"cohere-embed-v3-multilingual":{"id":"cohere-embed-v3-multilingual","name":"Embed v3 Multilingual","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"model-router":{"id":"model-router","name":"Model Router","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-05-19","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384},"cost":{"input":0.14,"output":0}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"llama-4-scout-17b-16e-instruct":{"id":"llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.78}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4-turbo-vision":{"id":"gpt-4-turbo-vision","name":"GPT-4 Turbo Vision","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":2.5,"output":10,"cache_read":1.25}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"status":"deprecated","cost":{"input":1.35,"output":5.4}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"phi-4-mini":{"id":"phi-4-mini","name":"Phi-4-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"phi-4-multimodal":{"id":"phi-4-multimodal","name":"Phi-4-multimodal","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"phi","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.08,"output":0.32,"input_audio":4}},"gpt-3.5-turbo-0125":{"id":"gpt-3.5-turbo-0125","name":"GPT-3.5 Turbo 0125","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"phi-4":{"id":"phi-4","name":"Phi-4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.125,"output":0.5}},"cohere-embed-v-4-0":{"id":"cohere-embed-v-4-0","name":"Embed v4","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":1536},"cost":{"input":0.12,"output":0}},"cohere-embed-v3-english":{"id":"cohere-embed-v3-english","name":"Embed v3 English","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.25,"output":1}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"cohere-command-a":{"id":"cohere-command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.5,"output":10}},"phi-4-reasoning-plus":{"id":"phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"codex-mini":{"id":"codex-mini","name":"Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-04","release_date":"2025-05-16","last_updated":"2025-05-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.5,"output":6,"cache_read":0.375}},"phi-4-mini-reasoning":{"id":"phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":15,"output":120}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-3.5-turbo-1106":{"id":"gpt-3.5-turbo-1106","name":"GPT-3.5 Turbo 1106","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":1,"output":2}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.04}},"mistral-small-2503":{"id":"mistral-small-2503","name":"Mistral Small 3.1","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}}}},"regolo-ai":{"id":"regolo-ai","env":["REGOLO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.regolo.ai/v1","name":"Regolo AI","doc":"https://docs.regolo.ai/","models":{"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5-9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":120000,"output":120000},"cost":{"input":0.58,"output":2.42}},"faster-whisper-large-v3":{"id":"faster-whisper-large-v3","name":"Faster Whisper Large v3","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0,"output":0}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":100000},"cost":{"input":0.46,"output":2.42}},"brick-complexity-pro":{"id":"brick-complexity-pro","name":"Brick Complexity Pro","description":"Complexity classifier that powers the Brick semantic router by extracting query difficulty","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":100000,"output":15000},"cost":{"input":0.12,"output":0.46}},"apertus-70b":{"id":"apertus-70b","name":"Apertus 70B","description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":30000,"output":30000},"cost":{"input":0.46,"output":2.42}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT-OSS-20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.4,"output":1.8}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3-Coder-Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.3,"output":1.2}},"qwen3-embedding-8b":{"id":"qwen3-embedding-8b","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.1,"output":0.1}},"qwen3-reranker-4b":{"id":"qwen3-reranker-4b","name":"Qwen3-Reranker-4B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.12,"output":0.12}},"glm5.2":{"id":"glm5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":96000,"output":96000},"cost":{"input":2.31,"output":6}},"brick-v1-beta":{"id":"brick-v1-beta","name":"Brick v1 Beta","description":"Semantic router by Regolo.ai that directs each request to the most suitable model, optimizing costs and performance","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":100000,"output":15000},"status":"beta","cost":{"input":0,"output":0}},"qwen-image":{"id":"qwen-image","name":"Qwen-Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.5,"output":2}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT-OSS-120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1,"output":4.2}},"deepseek-ocr-2":{"id":"deepseek-ocr-2","name":"DeepSeek OCR 2","description":"High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":4000,"output":4000},"cost":{"input":0,"output":0}},"mistral-small-4-119b":{"id":"mistral-small-4-119b","name":"Mistral Small 4 119B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.75,"output":3}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.7}},"qwen3.5-122b":{"id":"qwen3.5-122b","name":"Qwen3.5-122B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.9,"output":3.6}}}},"kenari":{"id":"kenari","env":["KENARI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://kenari.id/v1","name":"Kenari","doc":"https://kenari.id/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0}},"gpt-5-6-terra":{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"glm-5-1":{"id":"glm-5-1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"gemini-3-7-flash":{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-2-5-flash":{"id":"gemini-2-5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"kimi-k2-7-code:free":{"id":"kimi-k2-7-code:free","name":"Kimi K2.7 Code (Free)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"qwen3-7-plus":{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"deepseek-v4-flash:free":{"id":"deepseek-v4-flash:free","name":"DeepSeek V4 Flash (Free)","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gpt-5-6-sol":{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5-6-luna":{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0,"output":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"mistral-medium-3-5:free":{"id":"mistral-medium-3-5:free","name":"Mistral Medium 3.5 (Free)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"glm-5-3":{"id":"glm-5-3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"step-3-7-flash:free":{"id":"step-3-7-flash:free","name":"Step 3.7 Flash (Free)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"gemini-3-1-pro":{"id":"gemini-3-1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"grok-imagine-image-2-0":{"id":"grok-imagine-image-2-0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":8000,"output":0},"cost":{"input":0,"output":0}},"kimi-k2-6:free":{"id":"kimi-k2-6:free","name":"Kimi K2.6 (Free)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gemini-3-1-flash-lite":{"id":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"hy3:free":{"id":"hy3:free","name":"Hy3 (Free)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"mistral-large:free":{"id":"mistral-large:free","name":"Mistral Large (Free)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":0,"output":0}},"mimo-v2-5:free":{"id":"mimo-v2-5:free","name":"MiMo-V2.5 (Free)","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"nemotron-3-super-120b-a12b":{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"nemotron-3-super-120b-a12b:free":{"id":"nemotron-3-super-120b-a12b:free","name":"Nemotron 3 Super 120B A12B (Free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"whisper-large-v3-turbo":{"id":"whisper-large-v3-turbo","name":"Whisper Large v3 Turbo","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0,"output":0}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0}},"glm-5-2":{"id":"glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"glm-4-7-flash:free":{"id":"glm-4-7-flash:free","name":"GLM-4.7-Flash (Free)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mimo-v2-5":{"id":"mimo-v2-5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"gpt-image-2":{"id":"gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":272000,"output":16384},"cost":{"input":0,"output":0}},"mimo-v2-5-pro":{"id":"mimo-v2-5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":0,"output":0}},"step-3-7-flash":{"id":"step-3-7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gemini-2-5-flash-lite":{"id":"gemini-2-5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3-8-max":{"id":"qwen3-8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"nemotron-3-ultra-550b-a55b":{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemini-3-1-flash-tts":{"id":"gemini-3-1-flash-tts","name":"Gemini 3.1 Flash TTS Preview","description":"Low-latency speech generation with steerable prompts and expressive audio tags","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":0,"output":0}},"nemotron-3-nano-30b-a3b":{"id":"nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}}}},"the-grid-ai":{"id":"the-grid-ai","env":["THEGRID_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.thegrid.ai/v1","name":"The Grid AI","doc":"https://thegrid.ai/docs","models":{"agent-prime":{"id":"agent-prime","name":"Agent Prime","description":"Reliable models for dependable agentic applications, multi-step tool use, and reasoning workflows. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000},"status":"beta"},"text-standard":{"id":"text-standard","name":"Text Standard","description":"Price-optimized models with low-latency, high-throughput and shorter maximum outputs. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000}},"agent-max":{"id":"agent-max","name":"Agent Max","description":"Frontier models for autonomous research, deep multi-step tool chains, and complex long-horizon tasks. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-05-04","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"status":"beta"},"text-max":{"id":"text-max","name":"Text Max","description":"Frontier models for deep reasoning, long context, and complex workflows. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000}},"code-max":{"id":"code-max","name":"Code Max","description":"Frontier models for complex research, architectural decisions, debugging, and multi-file development. Any model that meets the contract spec can serve your request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-05-04","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"status":"beta"},"code-standard":{"id":"code-standard","name":"Code Standard","description":"Price-optimized models for rapid autocomplete, linting, high-frequency suggestions, and batch edits. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000},"status":"beta"},"code-prime":{"id":"code-prime","name":"Code Prime","description":"Reliable models for everyday software tasks, code completion, review, and standard debugging. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000},"status":"beta"},"text-prime":{"id":"text-prime","name":"Text Prime","description":"Reliable models for everyday text generation, editing, and analysis across diverse workflows. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"input":120000,"output":30000}},"agent-standard":{"id":"agent-standard","name":"Agent Standard","description":"Price-optimized models for fast tool calls, simple agent loops, high-throughput automation, and orchestration. Any model that meets the contract spec can serve your request.","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-04","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":120000,"output":16000},"status":"beta"}}},"google-vertex":{"id":"google-vertex","env":["GOOGLE_VERTEX_PROJECT","GOOGLE_VERTEX_LOCATION","GOOGLE_APPLICATION_CREDENTIALS"],"npm":"@ai-sdk/google-vertex","name":"Vertex","doc":"https://cloud.google.com/vertex-ai/generative-ai/docs/models","models":{"gemini-2.5-flash-tts":{"id":"gemini-2.5-flash-tts","name":"Gemini 2.5 Flash TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-09-30","last_updated":"2025-12-10","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.5,"output":10}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-pro-tts":{"id":"gemini-2.5-pro-tts","name":"Gemini 2.5 Pro TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-09-30","last_updated":"2025-12-10","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":1,"output":20}},"claude-sonnet-4@20250514":{"id":"claude-sonnet-4@20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"claude-opus-4-5@20251101":{"id":"claude-opus-4-5@20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":120,"cache_read":0.2}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"claude-sonnet-4-6@default":{"id":"claude-sonnet-4-6@default","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"claude-fable-5@default":{"id":"claude-fable-5@default","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-4-6@default":{"id":"claude-opus-4-6@default","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-4@20250514":{"id":"claude-opus-4@20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-haiku-4-5@20251001":{"id":"claude-haiku-4-5@20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-5@default":{"id":"claude-sonnet-5@default","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-4-1@20250805":{"id":"claude-opus-4-1@20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gemini-embedding-001":{"id":"gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60,"cache_read":0.05}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-fable-5-1@default":{"id":"claude-fable-5-1@default","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-7@default":{"id":"claude-opus-4-7@default","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"gemini-flash-lite-latest":{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"claude-opus-5@default":{"id":"claude-opus-5@default","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-flash-latest":{"id":"gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"claude-opus-4-8@default":{"id":"claude-opus-4-8@default","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-4-5@20250929":{"id":"claude-sonnet-4-5@20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/google-vertex/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen/qwen3-235b-a22b-instruct-2507-maas":{"id":"qwen/qwen3-235b-a22b-instruct-2507-maas","name":"Qwen3 235B A22B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.22,"output":0.88}},"deepseek-ai/deepseek-v3.1-maas":{"id":"deepseek-ai/deepseek-v3.1-maas","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":1.7,"cache_read":0.06}},"deepseek-ai/deepseek-v3.2-maas":{"id":"deepseek-ai/deepseek-v3.2-maas","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-17","last_updated":"2026-04-04","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.56,"output":1.68,"cache_read":0.056}},"zai-org/glm-5-maas":{"id":"zai-org/glm-5-maas","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1,"output":3.2,"cache_read":0.1}},"zai-org/glm-4.7-maas":{"id":"zai-org/glm-4.7-maas","name":"GLM-4.7","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-06","last_updated":"2026-01-06","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":2.2,"cache_read":0.06}},"meta/llama-4-maverick-17b-128e-instruct-maas":{"id":"meta/llama-4-maverick-17b-128e-instruct-maas","name":"Llama 4 Maverick 17B 128E Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":8192},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.35,"output":1.15}},"meta/llama-3.3-70b-instruct-maas":{"id":"meta/llama-3.3-70b-instruct-maas","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.72,"output":0.72}},"openai/gpt-oss-120b-maas":{"id":"openai/gpt-oss-120b-maas","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.36}},"openai/gpt-oss-20b-maas":{"id":"openai/gpt-oss-20b-maas","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"status":"deprecated","cost":{"input":0.07,"output":0.25,"cache_read":0.007}},"moonshotai/kimi-k2-thinking-maas":{"id":"moonshotai/kimi-k2-thinking-maas","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"xai/grok-4.20-reasoning":{"id":"xai/grok-4.20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":30000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-non-reasoning":{"id":"xai/grok-4.20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.1-fast-non-reasoning":{"id":"xai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast","description":"Fast Grok model for responsive chat, tool-assisted work, and low-latency responses","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":30000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.1-fast-reasoning":{"id":"xai/grok-4.1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":30000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":500000},"status":"beta","provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${GOOGLE_VERTEX_ENDPOINT}/v1/projects/${GOOGLE_VERTEX_PROJECT}/locations/${GOOGLE_VERTEX_LOCATION}/endpoints/openapi"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}}}},"stepfun-ai":{"id":"stepfun-ai","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.ai/v1","name":"StepFun (Global)","doc":"https://platform.stepfun.ai/docs/en/overview/concept","models":{"step-1-32k":{"id":"step-1-32k","name":"Step 1 (32K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":2.05,"output":9.59,"cache_read":0.41}},"stepaudio-2.5-tts":{"id":"stepaudio-2.5-tts","name":"StepAudio 2.5 TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"stepaudio-2.5-asr":{"id":"stepaudio-2.5-asr","name":"StepAudio 2.5 ASR","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-24","last_updated":"2026-07-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-2-16k":{"id":"step-2-16k","name":"Step 2 (16K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":5.21,"output":16.44,"cache_read":1.04}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-tts-2":{"id":"step-tts-2","name":"Step TTS 2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-06-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.185,"output":1.11,"cache_read":0.037}}}},"pendra":{"id":"pendra","env":["PENDRA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.pendra.ai/api/v1","name":"Pendra","doc":"https://pendra.ai/docs/integrations/opencode","models":{"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"gpt-oss:120b":{"id":"gpt-oss:120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3-coder:30b":{"id":"qwen3-coder:30b","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.6:27b":{"id":"qwen3.6:27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"llama3.3:70b":{"id":"llama3.3:70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}}}},"above":{"id":"above","env":["ABOVE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.above.dev/v1","name":"above.dev","doc":"https://above.dev/docs","models":{"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision (Exp)","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.242,"output":0.726,"reasoning":0.726,"cache_read":0.0077}},"glm-5.2":{"id":"glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.54,"output":4.84,"cache_read":0.154}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.242,"output":0.726,"reasoning":0.726,"cache_read":0.0077}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM 5.2 Fast","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.31,"output":7.26,"cache_read":0.231}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.165,"output":0.55,"cache_read":0.0319}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen 3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2.2,"output":6.6,"cache_read":0.275}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.726,"output":2.178,"reasoning":2.178,"cache_read":0.0242}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.5077,"output":1.0154,"cache_read":0.0042}}}},"scaleway":{"id":"scaleway","env":["SCALEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scaleway.ai/v1","name":"Scaleway","doc":"https://www.scaleway.com/en/docs/generative-apis/","models":{"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-01","last_updated":"2026-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"status":"beta","cost":{"input":0.25,"output":0.5}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"status":"beta","cost":{"input":0.468,"output":0.936,"reasoning":0.936,"cache_read":0.0936}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":1.8,"output":5.5}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.2,"output":0.8}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.6,"output":3.6}},"qwen3-embedding-8b":{"id":"qwen3-embedding-8b","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-05","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.1,"output":0}},"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2023-09","release_date":"2023-09-01","last_updated":"2026-03-17","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":8192},"cost":{"input":0.003,"output":0}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-01","last_updated":"2026-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"status":"beta","cost":{"input":0.25,"output":1.5}},"bge-multilingual-gemma2":{"id":"bge-multilingual-gemma2","name":"BGE Multilingual Gemma2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-26","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.1,"output":0}},"mistral-medium-3.5-128b":{"id":"mistral-medium-3.5-128b","name":"Mistral Medium 3.5 128B","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":1.5,"output":7.5}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral Small 3.2 24B Instruct (2506)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.35}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-01","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":260000,"output":16384},"cost":{"input":0.75,"output":2.25,"reasoning":8.4}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT-OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.6}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"Pixtral 12B 2409","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-25","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.2}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2026-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"output":16384},"cost":{"input":0.9,"output":0.9}}}},"alibaba-cn":{"id":"alibaba-cn","env":["DASHSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://dashscope.aliyuncs.com/compatible-mode/v1","name":"Alibaba (China)","doc":"https://www.alibabacloud.com/help/en/model-studio/models","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen2-5-72b-instruct":{"id":"qwen2-5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.574,"output":1.721}},"deepseek-r1-distill-qwen-7b":{"id":"deepseek-r1-distill-qwen-7b","name":"DeepSeek R1 Distill Qwen 7B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.072,"output":0.144}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":1.434}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.287,"output":1.147}},"qwen2-5-omni-7b":{"id":"qwen2-5-omni-7b","name":"Qwen2.5-Omni 7B","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0.087,"output":0.345,"input_audio":5.448}},"qwen-mt-turbo":{"id":"qwen-mt-turbo","name":"Qwen-MT Turbo","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.101,"output":0.28}},"deepseek-v3-1":{"id":"deepseek-v3-1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.574,"output":1.721}},"qwen-deep-research":{"id":"qwen-deep-research","name":"Qwen Deep Research","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":7.742,"output":23.367}},"qwen-vl-max":{"id":"qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.574}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":0.574}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.144,"output":0.574}},"qwen3-14b":{"id":"qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.574,"reasoning":1.434}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.345,"output":1.377}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"moonshot-kimi-k2-instruct":{"id":"moonshot-kimi-k2-instruct","name":"Moonshot Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.574,"output":2.294}},"qwen-vl-plus":{"id":"qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.115,"output":0.287}},"qwen-omni-turbo-realtime":{"id":"qwen-omni-turbo-realtime","name":"Qwen-Omni Turbo Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-05-08","last_updated":"2025-05-08","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.23,"output":0.918,"input_audio":3.584,"output_audio":7.168}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Moonshot Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.929,"output":3.858}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.022,"output":0.216}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.1,"output":3.851,"cache_read":0.275,"cache_write":0}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384},"cost":{"input":0.044,"output":0.087,"reasoning":0.431}},"qwen3-vl-30b-a3b":{"id":"qwen3-vl-30b-a3b","name":"Qwen3-VL 30B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.108,"output":0.431,"reasoning":1.076}},"qwen-vl-ocr":{"id":"qwen-vl-ocr","name":"Qwen-VL OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-28","last_updated":"2025-04-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":34096,"output":4096},"cost":{"input":0.717,"output":0.717}},"tongyi-intent-detect-v3":{"id":"tongyi-intent-detect-v3","name":"Tongyi Intent Detect V3","description":"General-purpose chat model for instruction following, writing, and analysis","family":"yi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1024},"cost":{"input":0.058,"output":0.144}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Moonshot Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.574,"output":2.294}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.287,"output":1.147,"reasoning":2.868}},"qwq-plus":{"id":"qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.574}},"qwen3.5-flash":{"id":"qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.172,"output":1.72,"reasoning":1.72}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.143353,"output":1.433525,"reasoning":4.300576}},"deepseek-v3-2-exp":{"id":"deepseek-v3-2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.287,"output":0.431}},"qwen2-5-7b-instruct":{"id":"qwen2-5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.072,"output":0.144}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.216,"output":0.861,"tiers":[{"input":0.323,"output":1.291,"tier":{"type":"context","size":32000}},{"input":0.538,"output":2.151,"tier":{"type":"context","size":128000}}]}},"qwen2-5-math-7b-instruct":{"id":"qwen2-5-math-7b-instruct","name":"Qwen2.5-Math 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3072},"cost":{"input":0.144,"output":0.287}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.172,"output":1.032,"reasoning":1.032,"tiers":[{"input":0.43,"output":2.58,"reasoning":2.58,"tier":{"type":"context","size":128000}}]}},"qwq-32b":{"id":"qwq-32b","name":"QwQ 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.574,"output":2.294}},"qwen3-asr-flash":{"id":"qwen3-asr-flash","name":"Qwen3-ASR Flash","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-04","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":0.032,"output":0.032}},"qwen3-omni-flash":{"id":"qwen3-omni-flash","name":"Qwen3-Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.058,"output":0.23,"input_audio":3.584,"output_audio":7.168}},"deepseek-r1-distill-qwen-1-5b":{"id":"deepseek-r1-distill-qwen-1-5b","name":"DeepSeek R1 Distill Qwen 1.5B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.02962,"output":0.1185,"cache_read":0.002962,"cache_write":0.03703,"tiers":[{"input":0.08887,"output":0.35549,"cache_read":0.008887,"cache_write":0.11109,"tier":{"type":"context","size":32000}},{"input":0.17774,"output":0.71098,"cache_read":0.017774,"cache_write":0.22218,"tier":{"type":"context","size":256000}}]}},"qwen2-5-vl-72b-instruct":{"id":"qwen2-5-vl-72b-instruct","name":"Qwen2.5-VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.294,"output":6.881}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.574,"output":2.294}},"deepseek-r1-distill-qwen-32b":{"id":"deepseek-r1-distill-qwen-32b","name":"DeepSeek R1 Distill Qwen 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.287,"output":0.861}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.861,"output":3.441}},"qwen2-5-coder-32b-instruct":{"id":"qwen2-5-coder-32b-instruct","name":"Qwen2.5-Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11","last_updated":"2024-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.287,"output":0.861}},"qwen2-5-math-72b-instruct":{"id":"qwen2-5-math-72b-instruct","name":"Qwen2.5-Math 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3072},"cost":{"input":0.574,"output":1.721}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.115,"output":0.287,"reasoning":1.147,"cache_read":0.012,"cache_write":0.144}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"qwen3-omni-flash-realtime":{"id":"qwen3-omni-flash-realtime","name":"Qwen3-Omni Flash Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.23,"output":0.918,"input_audio":3.584,"output_audio":7.168}},"qwen2-5-vl-7b-instruct":{"id":"qwen2-5-vl-7b-instruct","name":"Qwen2.5-VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.717}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.11875,"output":0.40073,"cache_read":0.01187,"cache_write":0.14844}},"qwen2-5-14b-instruct":{"id":"qwen2-5-14b-instruct","name":"Qwen2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.431}},"qwen-math-turbo":{"id":"qwen-math-turbo","name":"Qwen Math Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3072},"cost":{"input":0.287,"output":0.861}},"qwen-plus-character":{"id":"qwen-plus-character","name":"Qwen Plus Character","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.115,"output":0.287}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":245800,"output":65536},"cost":{"input":1.32,"output":7.9,"cache_read":0.132}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0.573,"output":2.58,"tiers":[{"input":0.86,"output":3.154,"tier":{"type":"context","size":32000}}]}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.77744,"output":5.33231,"cache_read":0.22218,"cache_write":2.22179}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Moonshot Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.574,"output":2.411}},"qwen3-8b":{"id":"qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.072,"output":0.287,"reasoning":0.717}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0.825,"output":3.301,"cache_read":0.17,"tiers":[{"input":1.1,"output":3.851,"tier":{"type":"context","size":32000}}]}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.287,"output":1.147,"reasoning":2.868}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":128000}}]}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.058,"output":0.23,"input_audio":3.584,"output_audio":7.168}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.861,"output":3.441,"tiers":[{"input":1.291,"output":5.161,"tier":{"type":"context","size":32000}},{"input":2.151,"output":8.602,"tier":{"type":"context","size":128000}}]}},"qwen2-5-32b-instruct":{"id":"qwen2-5-32b-instruct","name":"Qwen2.5 32B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.287,"output":0.861}},"qwen2-5-coder-7b-instruct":{"id":"qwen2-5-coder-7b-instruct","name":"Qwen2.5-Coder 7B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11","last_updated":"2024-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.144,"output":0.287}},"qwen-long":{"id":"qwen-long","name":"Qwen Long","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-25","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":10000000,"output":8192},"cost":{"input":0.072,"output":0.287}},"deepseek-r1-distill-llama-8b":{"id":"deepseek-r1-distill-llama-8b","name":"DeepSeek R1 Distill Llama 8B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"deepseek-r1-distill-qwen-14b":{"id":"deepseek-r1-distill-qwen-14b","name":"DeepSeek R1 Distill Qwen 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.144,"output":0.431}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3-VL 235B-A22B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.286705,"output":1.14682,"reasoning":2.867051}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.573,"output":3.44,"reasoning":3.44}},"qwen-math-plus":{"id":"qwen-math-plus","name":"Qwen Math Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-08-16","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3072},"cost":{"input":0.574,"output":1.721}},"qwen-doc-turbo":{"id":"qwen-doc-turbo","name":"Qwen Doc Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.087,"output":0.144}},"qwen-mt-plus":{"id":"qwen-mt-plus","name":"Qwen-MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.259,"output":0.775}},"qvq-max":{"id":"qvq-max","name":"QVQ Max","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qvq","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":1.147,"output":4.588}},"MiniMax/MiniMax-M2.7":{"id":"MiniMax/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"siliconflow/deepseek-r1-0528":{"id":"siliconflow/deepseek-r1-0528","name":"siliconflow/deepseek-r1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.5,"output":2.18}},"siliconflow/deepseek-v3.2":{"id":"siliconflow/deepseek-v3.2","name":"siliconflow/deepseek-v3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.42}},"siliconflow/deepseek-v3.1-terminus":{"id":"siliconflow/deepseek-v3.1-terminus","name":"siliconflow/deepseek-v3.1-terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":1}},"siliconflow/deepseek-v3-0324":{"id":"siliconflow/deepseek-v3-0324","name":"siliconflow/deepseek-v3-0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":1}},"kimi/kimi-k2.5":{"id":"kimi/kimi-k2.5","name":"kimi/kimi-k2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":5}}}},"poe":{"id":"poe","env":["POE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.poe.com/v1","name":"Poe","doc":"https://creator.poe.com/docs/external-applications/openai-compatible-api","models":{"poetools/claude-code":{"id":"poetools/claude-code","name":"claude-code","description":"Claude model for careful reasoning, writing, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-27","last_updated":"2025-11-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"elevenlabs/elevenlabs-v2.5-turbo":{"id":"elevenlabs/elevenlabs-v2.5-turbo","name":"ElevenLabs-v2.5-Turbo","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-28","last_updated":"2024-10-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":128000,"output":0}},"elevenlabs/elevenlabs-v3":{"id":"elevenlabs/elevenlabs-v3","name":"ElevenLabs-v3","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":128000,"output":0}},"elevenlabs/elevenlabs-music":{"id":"elevenlabs/elevenlabs-music","name":"ElevenLabs-Music","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"elevenlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-29","last_updated":"2025-08-29","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":2000,"output":0}},"stabilityai/stablediffusionxl":{"id":"stabilityai/stablediffusionxl","name":"StableDiffusionXL","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"stable-diffusion","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-07-09","last_updated":"2023-07-09","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":200,"output":0}},"trytako/tako":{"id":"trytako/tako","name":"Tako","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"tako","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":0}},"ideogramai/ideogram-v2a":{"id":"ideogramai/ideogram-v2a","name":"Ideogram-v2a","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram-v2a-turbo":{"id":"ideogramai/ideogram-v2a-turbo","name":"Ideogram-v2a-Turbo","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram-v2":{"id":"ideogramai/ideogram-v2","name":"Ideogram-v2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-21","last_updated":"2024-08-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"ideogramai/ideogram":{"id":"ideogramai/ideogram","name":"Ideogram","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ideogram","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-04-03","last_updated":"2024-04-03","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":150,"output":0}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude-Opus-4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":4.2929,"output":21.4646}},"anthropic/claude-sonnet-3.5":{"id":"anthropic/claude-sonnet-3.5","name":"Claude-Sonnet-3.5","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-06-05","last_updated":"2024-06-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"status":"deprecated","cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude-Opus-4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":128000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.4}},"anthropic/claude-sonnet-3.7":{"id":"anthropic/claude-sonnet-3.7","name":"Claude-Sonnet-3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":128000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude-Opus-4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":31999}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":32000},"cost":{"input":13,"output":64,"cache_read":1.3,"cache_write":16}},"anthropic/claude-haiku-3":{"id":"anthropic/claude-haiku-3","name":"Claude-Haiku-3","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-03-09","last_updated":"2024-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"cost":{"input":0.21,"output":1.1,"cache_read":0.021,"cache_write":0.26}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude-Sonnet-4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":128000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-haiku-3.5":{"id":"anthropic/claude-haiku-3.5","name":"Claude-Haiku-3.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"cost":{"input":0.68,"output":3.4,"cache_read":0.068,"cache_write":0.85}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude-Haiku-4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":63999}],"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":64000},"cost":{"input":0.85,"output":4.3,"cache_read":0.085,"cache_write":1.1}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude-Opus-4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":128000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.3}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude-Opus-4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":192512,"output":28672},"cost":{"input":13,"output":64,"cache_read":1.3,"cache_write":16}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude-Sonnet-4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":31999}],"tool_call":true,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":32768},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-sonnet-3.5-june":{"id":"anthropic/claude-sonnet-3.5-june","name":"Claude-Sonnet-3.5-June","description":"Legacy model retained for compatibility with older integrations","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":189096,"output":8192},"status":"deprecated","cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude-Opus-4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":0,"max":63999}],"tool_call":true,"temperature":false,"release_date":"2025-11-21","last_updated":"2025-11-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":196608,"output":64000},"cost":{"input":4.3,"output":21,"cache_read":0.43,"cache_write":5.3}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude-Sonnet-4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":983040,"output":64000},"cost":{"input":2.6,"output":13,"cache_read":0.26,"cache_write":3.2}},"google/nano-banana-pro":{"id":"google/nano-banana-pro","name":"Nano-Banana-Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"nano-banana","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":65536,"output":0},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-3.1-pro":{"id":"google/gemini-3.1-pro","name":"Gemini-3.1-Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-deep-research":{"id":"google/gemini-deep-research","name":"gemini-deep-research","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":0},"status":"deprecated","cost":{"input":1.6,"output":9.6}},"google/gemini-2.0-flash":{"id":"google/gemini-2.0-flash","name":"Gemini-2.0-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":990000,"output":8192},"cost":{"input":0.1,"output":0.42}},"google/veo-3.1-fast":{"id":"google/veo-3.1-fast","name":"Veo-3.1-Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/nano-banana":{"id":"google/nano-banana","name":"Nano-Banana","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"nano-banana","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":0},"cost":{"input":0.21,"output":1.8,"cache_read":0.021}},"google/imagen-4":{"id":"google/imagen-4","name":"Imagen-4","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini-2.5-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":false,"release_date":"2025-06-19","last_updated":"2025-06-19","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":64000},"cost":{"input":0.07,"output":0.28}},"google/imagen-3-fast":{"id":"google/imagen-3-fast","name":"Imagen-3-Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-17","last_updated":"2024-10-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.0-flash-lite":{"id":"google/gemini-2.0-flash-lite","name":"Gemini-2.0-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":990000,"output":8192},"cost":{"input":0.052,"output":0.21}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini-3.1-Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"google/veo-3.1":{"id":"google/veo-3.1","name":"Veo-3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/veo-3-fast":{"id":"google/veo-3-fast","name":"Veo-3-Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-13","last_updated":"2025-10-13","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/imagen-4-fast":{"id":"google/imagen-4-fast","name":"Imagen-4-Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini-3.5-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5152,"output":9.0909,"cache_read":0.1515}},"google/veo-3":{"id":"google/veo-3","name":"Veo-3","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-21","last_updated":"2025-05-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3-pro":{"id":"google/gemini-3-pro","name":"Gemini-3-Pro","description":"Legacy model retained for compatibility with older integrations","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":1.6,"output":9.6,"cache_read":0.16}},"google/lyria":{"id":"google/lyria","name":"Lyria","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-06-04","last_updated":"2025-06-04","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemma-4-31b":{"id":"google/gemma-4-31b","name":"Gemma-4-31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"google/imagen-4-ultra":{"id":"google/imagen-4-ultra","name":"Imagen-4-Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-24","last_updated":"2025-05-24","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/veo-2":{"id":"google/veo-2","name":"Veo-2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini-2.5-Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":32768}],"tool_call":true,"temperature":false,"release_date":"2025-02-05","last_updated":"2025-02-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1065535,"output":65535},"cost":{"input":0.87,"output":7,"cache_read":0.087}},"google/gemini-3-flash":{"id":"google/gemini-3-flash","name":"Gemini-3-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini-2.5-Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":false,"release_date":"2025-04-26","last_updated":"2025-04-26","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1065535,"output":65535},"cost":{"input":0.21,"output":1.8,"cache_read":0.021}},"google/imagen-3":{"id":"google/imagen-3","name":"Imagen-3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-15","last_updated":"2024-10-15","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"novita/glm-4.7":{"id":"novita/glm-4.7","name":"glm-4.7","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072},"status":"deprecated"},"novita/glm-4.6":{"id":"novita/glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"novita/minimax-m2.1":{"id":"novita/minimax-m2.1","name":"minimax-m2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-26","last_updated":"2025-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072}},"novita/glm-4.6v":{"id":"novita/glm-4.6v","name":"glm-4.6v","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":32768}},"novita/kimi-k2.6":{"id":"novita/kimi-k2.6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-05-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.96,"output":4.04,"cache_read":0.16}},"novita/kimi-k2-thinking":{"id":"novita/kimi-k2-thinking","name":"kimi-k2-thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-11-07","last_updated":"2025-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":0}},"novita/deepseek-v3.2":{"id":"novita/deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":0},"cost":{"input":0.27,"output":0.4,"cache_read":0.13}},"novita/glm-4.7-n":{"id":"novita/glm-4.7-n","name":"glm-4.7-n","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072}},"novita/glm-5":{"id":"novita/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":205000,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"novita/kimi-k2.5":{"id":"novita/kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"novita/glm-4.7-flash":{"id":"novita/glm-4.7-flash","name":"glm-4.7-flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65500}},"fireworks-ai/kimi-k2.5-fw":{"id":"fireworks-ai/kimi-k2.5-fw","name":"Kimi-K2.5-FW","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"input":245760,"output":16384},"cost":{"input":0,"output":0}},"lumalabs/ray2":{"id":"lumalabs/ray2","name":"Ray2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ray","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-02-20","last_updated":"2025-02-20","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":5000,"output":0}},"empiriolabs/deepseek-v4-flash-el":{"id":"empiriolabs/deepseek-v4-flash-el","name":"DeepSeek-V4-Flash-EL","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-05-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":0.14,"output":0.28}},"empiriolabs/deepseek-v4-pro-el":{"id":"empiriolabs/deepseek-v4-pro-el","name":"DeepSeek-V4-Pro-EL","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-04-24","last_updated":"2026-05-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":384000},"cost":{"input":1.67,"output":3.33}},"topazlabs-co/topazlabs":{"id":"topazlabs-co/topazlabs","name":"TopazLabs","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"topazlabs","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":204,"output":0}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.045,"output":0.36,"cache_read":0.0045}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.09,"output":0.36,"cache_read":0.022}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":14,"output":110}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"o3-mini-high","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-01-31","last_updated":"2025-01-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1-Codex-Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/chatgpt-4o-latest":{"id":"openai/chatgpt-4o-latest","name":"ChatGPT-4o-Latest","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-14","last_updated":"2024-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"status":"deprecated","cost":{"input":4.5,"output":14}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-4-classic":{"id":"openai/gpt-4-classic","name":"GPT-4-Classic","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-03-25","last_updated":"2024-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"status":"deprecated","cost":{"input":27,"output":54}},"openai/o3-deep-research":{"id":"openai/o3-deep-research","name":"o3-deep-research","description":"Research model for long-horizon investigation, synthesis, and analytical reports","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":9,"output":36,"cache_read":2.2}},"openai/sora-2":{"id":"openai/sora-2","name":"Sora-2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"sora","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.3-instant":{"id":"openai/gpt-5.3-instant","name":"GPT-5.3-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":19,"output":150}},"openai/gpt-5.3-codex-spark":{"id":"openai/gpt-5.3-codex-spark","name":"GPT-5.3-Codex-Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.36,"output":1.4,"cache_read":0.09}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":14,"cache_read":0.22}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4-Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-13","last_updated":"2023-09-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":9,"output":27}},"openai/gpt-3.5-turbo-raw":{"id":"openai/gpt-3.5-turbo-raw","name":"GPT-3.5-Turbo-Raw","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-27","last_updated":"2023-09-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":4524,"output":2048},"cost":{"input":0.45,"output":1.4}},"openai/gpt-5.2-instant":{"id":"openai/gpt-5.2-instant","name":"GPT-5.2-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/dall-e-3":{"id":"openai/dall-e-3","name":"DALL-E-3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"dall-e","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":800,"output":0}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1-Codex-Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/o4-mini-deep-research":{"id":"openai/o4-mini-deep-research","name":"o4-mini-deep-research","description":"Research model for long-horizon investigation, synthesis, and analytical reports","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":14,"output":54}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-02-10","last_updated":"2026-02-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":124096,"output":4096},"cost":{"input":0.14,"output":0.54,"cache_read":0.068}},"openai/gpt-image-1.5":{"id":"openai/gpt-image-1.5","name":"gpt-image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":140,"output":540}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/gpt-image-1":{"id":"openai/gpt-image-1","name":"GPT-Image-1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4-Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.18,"output":1.1,"cache_read":0.018}},"openai/gpt-4o-search":{"id":"openai/gpt-4o-search","name":"GPT-4o-Search","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-11","last_updated":"2025-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2.2,"output":9}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5-Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":27.2727,"output":163.6364}},"openai/gpt-image-1-mini":{"id":"openai/gpt-image-1-mini","name":"GPT-Image-1-Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-4o-aug":{"id":"openai/gpt-4o-aug","name":"GPT-4o-Aug","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-11-21","last_updated":"2024-11-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2.2,"output":9,"cache_read":1.1}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4-Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-12","last_updated":"2026-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.68,"output":4,"cache_read":0.068}},"openai/gpt-5.1-instant":{"id":"openai/gpt-5.1-instant","name":"GPT-5.1-Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5.0505,"output":32.3232,"cache_read":1.2626}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-13","last_updated":"2023-09-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":2048},"cost":{"input":0.45,"output":1.4}},"openai/gpt-5-chat":{"id":"openai/gpt-5-chat","name":"GPT-5-Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.22,"output":1.8,"cache_read":0.022}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4-Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":27,"output":160}},"openai/sora-2-pro":{"id":"openai/sora-2-pro","name":"Sora-2-Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"sora","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-4o-mini-search":{"id":"openai/gpt-4o-mini-search","name":"GPT-4o-mini-Search","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-03-11","last_updated":"2025-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.14,"output":0.54}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5-Turbo-Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2023-09-20","last_updated":"2023-09-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":3500,"output":1024},"cost":{"input":1.4,"output":1.8}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.6,"output":13,"cache_read":0.16}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.1,"output":9,"cache_read":0.11}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4,"cache_read":0.25}},"openai/gpt-4-classic-0314":{"id":"openai/gpt-4-classic-0314","name":"GPT-4-Classic-0314","description":"Legacy model retained for compatibility with older integrations","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-08-26","last_updated":"2024-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"status":"deprecated","cost":{"input":27,"output":54}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-01-31","last_updated":"2025-01-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.99,"output":4}},"openai/o3":{"id":"openai/o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.8,"output":7.2,"cache_read":0.45}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":18,"output":72}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":4.5455,"output":27.2727,"cache_read":0.4545}},"xai/grok-3-mini":{"id":"xai/grok-3-mini","name":"Grok 3 Mini","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-04-11","last_updated":"2025-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.3,"output":0.5,"cache_read":0.075}},"xai/grok-4.20-multi-agent":{"id":"xai/grok-4.20-multi-agent","name":"Grok-4.20-Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2026-03-13","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-code-fast-1":{"id":"xai/grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-08-22","last_updated":"2025-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"xai/grok-4":{"id":"xai/grok-4","name":"Grok-4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-3":{"id":"xai/grok-3","name":"Grok 3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-04-11","last_updated":"2025-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.75}},"xai/grok-4-fast-reasoning":{"id":"xai/grok-4-fast-reasoning","name":"Grok-4-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-09-16","last_updated":"2025-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.1-fast-non-reasoning":{"id":"xai/grok-4.1-fast-non-reasoning","name":"Grok-4.1-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000}},"xai/grok-4.1-fast-reasoning":{"id":"xai/grok-4.1-fast-reasoning","name":"Grok-4.1-Fast-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000}},"xai/grok-4-fast-non-reasoning":{"id":"xai/grok-4-fast-non-reasoning","name":"Grok-4-Fast-Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-09-16","last_updated":"2025-09-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"cerebras/qwen3-32b-cs":{"id":"cerebras/qwen3-32b-cs","name":"qwen3-32b-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-05-15","last_updated":"2025-05-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"cerebras/llama-3.1-8b-cs":{"id":"cerebras/llama-3.1-8b-cs","name":"Llama-3.1-8B-CS","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-13","last_updated":"2025-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":0.1,"output":0.1}},"cerebras/llama-3.3-70b-cs":{"id":"cerebras/llama-3.3-70b-cs","name":"llama-3.3-70b-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-13","last_updated":"2025-05-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"cerebras/gpt-oss-120b-cs":{"id":"cerebras/gpt-oss-120b-cs","name":"GPT-OSS-120B-CS","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":0},"cost":{"input":0.35,"output":0.75}},"cerebras/qwen3-235b-2507-cs":{"id":"cerebras/qwen3-235b-2507-cs","name":"qwen3-235b-2507-cs","description":"Legacy model retained for compatibility with older integrations","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"status":"deprecated"},"runwayml/runway-gen-4-turbo":{"id":"runwayml/runway-gen-4-turbo","name":"Runway-Gen-4-Turbo","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"runway","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2025-05-09","last_updated":"2025-05-09","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":256,"output":0}},"runwayml/runway":{"id":"runwayml/runway","name":"Runway","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"runway","attachment":true,"reasoning":false,"tool_call":true,"temperature":false,"release_date":"2024-10-11","last_updated":"2024-10-11","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":256,"output":0}}}},"modelscope":{"id":"modelscope","env":["MODELSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-inference.modelscope.cn/v1","name":"ModelScope","doc":"https://modelscope.cn/docs/model-service/API-Inference/intro","models":{"ZhipuAI/GLM-4.5":{"id":"ZhipuAI/GLM-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0}},"ZhipuAI/GLM-4.6":{"id":"ZhipuAI/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":98304},"cost":{"input":0,"output":0}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"Qwen/Qwen3-30B-A3B-Thinking-2507":{"id":"Qwen/Qwen3-30B-A3B-Thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}}}},"poolside":{"id":"poolside","env":["POOLSIDE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.poolside.ai/v1","name":"Poolside","doc":"https://platform.poolside.ai","models":{"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"poolside/laguna-m.1":{"id":"poolside/laguna-m.1","name":"Laguna M.1","description":"Poolside's open-weight model for agentic coding and long-horizon work","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"claudinio":{"id":"claudinio","env":["CLAUDINIO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.claudin.io/v1","name":"Claudinio","doc":"https://claudin.io","models":{"claudinio":{"id":"claudinio","name":"Claudinio","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2026-05","release_date":"2026-05-12","last_updated":"2026-06-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.5,"output":2,"cache_read":0.15}},"claudius":{"id":"claudius","name":"Claudius","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"knowledge":"2026-05","release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":3,"output":8,"cache_read":0.9}}}},"novita-ai":{"id":"novita-ai","env":["NOVITA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.novita.ai/openai","name":"NovitaAI","doc":"https://novita.ai/docs/guides/introduction","models":{"paddlepaddle/paddleocr-vl":{"id":"paddlepaddle/paddleocr-vl","name":"PaddleOCR-VL","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-22","last_updated":"2025-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.02}},"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7-Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.5625}},"qwen/qwen3-omni-30b-a3b-instruct":{"id":"qwen/qwen3-omni-30b-a3b-instruct","name":"Qwen3 Omni 30B A3B Instruct","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","video","audio","image"],"output":["text","audio"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.25,"output":0.97,"input_audio":2.2,"output_audio":1.788}},"qwen/qwen3-30b-a3b-fp8":{"id":"qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.09,"output":0.45}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22b Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":3}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-10","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5-27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5-35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"qwen/qwen3-235b-a22b-fp8":{"id":"qwen/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"qwen/qwen3-4b-fp8":{"id":"qwen/qwen3-4b-fp8","name":"Qwen3 4B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":20000},"cost":{"input":0.03,"output":0.03}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.8,"output":0.8}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30b A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.07,"output":0.27}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":2.11,"output":8.45}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"qwen/qwen3-vl-8b-instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-17","last_updated":"2025-10-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.08,"output":0.5}},"qwen/qwen3-8b-fp8":{"id":"qwen/qwen3-8b-fp8","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":20000},"cost":{"input":0.035,"output":0.138}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"qwen/qwen3-vl-30b-a3b-instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.7}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5-122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"qwen/qwen3-32b-fp8":{"id":"qwen/qwen3-32b-fp8","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.1,"output":0.45}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"qwen/qwen3-vl-30b-a3b-thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-10-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":1}},"qwen/qwen2.5-7b-instruct":{"id":"qwen/qwen2.5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.07,"output":0.07}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen 2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-15","last_updated":"2024-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.38,"output":0.4}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.09,"output":0.58}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"qwen/qwen3-omni-30b-a3b-thinking":{"id":"qwen/qwen3-omni-30b-a3b-thinking","name":"Qwen3 Omni 30B A3B Thinking","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text","audio","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.25,"output":0.97,"input_audio":2.2,"output_audio":1.788}},"qwen/qwen-mt-plus":{"id":"qwen/qwen-mt-plus","name":"Qwen MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-03","last_updated":"2025-09-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":8192},"cost":{"input":0.25,"output":0.75}},"baidu/ernie-4.5-300b-a47b-paddle":{"id":"baidu/ernie-4.5-300b-a47b-paddle","name":"ERNIE 4.5 300B A47B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":12000},"cost":{"input":0.28,"output":1.1}},"baidu/ernie-4.5-vl-28b-a3b":{"id":"baidu/ernie-4.5-vl-28b-a3b","name":"ERNIE 4.5 VL 28B A3B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2026-06-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":30000,"output":8000},"cost":{"input":0.14,"output":0.56}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"baidu/ernie-4.5-21B-a3b-thinking":{"id":"baidu/ernie-4.5-21B-a3b-thinking","name":"ERNIE-4.5-21B-A3B-Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.07,"output":0.28}},"baidu/ernie-4.5-21B-a3b":{"id":"baidu/ernie-4.5-21B-a3b","name":"ERNIE 4.5 21B A3B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":120000,"output":8000},"cost":{"input":0.07,"output":0.28}},"baidu/ernie-4.5-vl-28b-a3b-thinking":{"id":"baidu/ernie-4.5-vl-28b-a3b-thinking","name":"ERNIE-4.5-VL-28B-A3B-Thinking","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-26","last_updated":"2025-11-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.39,"output":0.39}},"kwaipilot/kat-coder-pro":{"id":"kwaipilot/kat-coder-pro","name":"Kat Coder Pro","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-05","last_updated":"2026-01-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-07-30","last_updated":"2024-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60288,"output":16000},"cost":{"input":0.04,"output":0.17}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131100},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax-m2.5","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.03}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.13,"output":0.4}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":98304,"output":16384},"cost":{"input":0.119,"output":0.2}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.05,"output":0.1}},"zai-org/glm-4.7":{"id":"zai-org/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.5-air":{"id":"zai-org/glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-10-13","last_updated":"2025-10-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"zai-org/glm-4.6":{"id":"zai-org/glm-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.6v":{"id":"zai-org/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/autoglm-phone-9b-multilingual":{"id":"zai-org/autoglm-phone-9b-multilingual","name":"AutoGLM-Phone-9B-Multilingual","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.035,"output":0.138}},"zai-org/glm-4.5":{"id":"zai-org/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/glm-4.5v":{"id":"zai-org/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","video","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai-org/glm-5":{"id":"zai-org/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/glm-5.1":{"id":"zai-org/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.38,"output":4.4,"cache_read":0.26}},"zai-org/glm-4.7-flash":{"id":"zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4,"cache_read":0.01}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"Mythomax L2 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":3200},"cost":{"input":0.09,"output":0.09}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"Wizardlm 2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-24","last_updated":"2024-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"minimaxai/minimax-m1-80k":{"id":"minimaxai/minimax-m1-80k","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"deepseek/deepseek-ocr":{"id":"deepseek/deepseek-ocr","name":"DeepSeek-OCR","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-24","last_updated":"2025-10-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.03,"output":0.03}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-prover-v2-671b":{"id":"deepseek/deepseek-prover-v2-671b","name":"Deepseek Prover V2 671B","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":160000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"Deepseek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-r1-distill-qwen-32b":{"id":"deepseek/deepseek-r1-distill-qwen-32b","name":"DeepSeek R1 Distill Qwen 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":32000},"cost":{"input":0.3,"output":0.3}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill LLama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-r1-turbo":{"id":"deepseek/deepseek-r1-turbo","name":"DeepSeek R1 (Turbo)\t","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-r1-0528-qwen3-8b":{"id":"deepseek/deepseek-r1-0528-qwen3-8b","name":"DeepSeek R1 0528 Qwen3 8B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.06,"output":0.09}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"Deepseek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"Deepseek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v3-turbo":{"id":"deepseek/deepseek-v3-turbo","name":"DeepSeek V3 (Turbo)\t","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.4,"output":1.3}},"deepseek/deepseek-v3-0324":{"id":"deepseek/deepseek-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.27,"output":1.12,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.6,"output":3.2,"cache_read":0.135}},"deepseek/deepseek-ocr-2":{"id":"deepseek/deepseek-ocr-2","name":"deepseek/deepseek-ocr-2","description":"OCR model for extracting structured text from documents and screenshots","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.03,"output":0.03}},"deepseek/deepseek-r1-distill-qwen-14b":{"id":"deepseek/deepseek-r1-distill-qwen-14b","name":"DeepSeek R1 Distill Qwen 14B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.15,"output":0.15}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"inclusionai/ring-2.6-1t":{"id":"inclusionai/ring-2.6-1t","name":"Ring-2.6-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-08","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ling-2.6-1t":{"id":"inclusionai/ling-2.6-1t","name":"Ling-2.6-1T","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-23","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ling-2.6-flash":{"id":"inclusionai/ling-2.6-flash","name":"Ling-2.6-flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"xiaomimimo/mimo-v2-flash":{"id":"xiaomimimo/mimo-v2-flash","name":"XiaomiMiMo/MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-19","last_updated":"2025-12-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.1,"output":0.3,"cache_read":0.3}},"xiaomimimo/mimo-v2-pro":{"id":"xiaomimimo/mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.4,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomimimo/mimo-v2.5-pro":{"id":"xiaomimimo/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.522,"output":1.044,"cache_read":0.0043,"tiers":[{"input":0.522,"output":1.044,"cache_read":0.0043,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.522,"output":1.044,"cache_read":0.0043}}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-07-24","last_updated":"2024-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.05}},"meta-llama/llama-4-maverick-17b-128e-instruct-fp8":{"id":"meta-llama/llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-06","last_updated":"2025-04-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.27,"output":0.85}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"meta-llama/llama-3-8b-instruct":{"id":"meta-llama/llama-3-8b-instruct","name":"Llama 3 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.04,"output":0.04}},"meta-llama/llama-4-scout-17b-16e-instruct":{"id":"meta-llama/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-06","last_updated":"2025-04-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.18,"output":0.59}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-07","last_updated":"2024-12-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":120000},"cost":{"input":0.135,"output":0.4}},"meta-llama/llama-3-70b-instruct":{"id":"meta-llama/llama-3-70b-instruct","name":"Llama3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-04-25","last_updated":"2024-04-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"nousresearch/hermes-2-pro-llama-3-8b":{"id":"nousresearch/hermes-2-pro-llama-3-8b","name":"Hermes 2 Pro Llama 3 8B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-06-27","last_updated":"2024-06-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.14,"output":0.14}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"OpenAI: GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.15}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-06","last_updated":"2025-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.25}},"sao10K/l3-70b-euryale-v2.1":{"id":"sao10K/l3-70b-euryale-v2.1","name":"L3 70B Euryale V2.1\t","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-06-18","last_updated":"2024-06-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":1.48,"output":1.48}},"sao10K/l3-8b-lunaris":{"id":"sao10K/l3-8b-lunaris","name":"Sao10k L3 8B Lunaris\t","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-11-28","last_updated":"2024-11-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.05,"output":0.05}},"sao10K/L3-8B-stheno-v3.2":{"id":"sao10K/L3-8B-stheno-v3.2","name":"L3 8B Stheno V3.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-11-29","last_updated":"2024-11-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":32000},"cost":{"input":0.05,"output":0.05}},"sao10K/l31-70b-euryale-v2.2":{"id":"sao10K/l31-70b-euryale-v2.2","name":"L31 70B Euryale V2.2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":1.48,"output":1.48}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-11-07","last_updated":"2026-06-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"baichuan/baichuan-m2-32b":{"id":"baichuan/baichuan-m2-32b","name":"baichuan-m2-32b","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"baichuan","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.07,"output":0.07}}}},"nebius":{"id":"nebius","env":["NEBIUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokenfactory.nebius.com/v1","name":"Nebius Token Factory","doc":"https://docs.tokenfactory.nebius.com/","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.15}},"nvidia/Nemotron-3_5-Lightning":{"id":"nvidia/Nemotron-3_5-Lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"nvidia/Nemotron-3-Ultra-550b-a55b":{"id":"nvidia/Nemotron-3-Ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":3,"cache_read":1}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron-3-Super-120B-A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":32768},"cost":{"input":0.3,"output":0.9}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma-3-27b-it","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-10","release_date":"2026-01-20","last_updated":"2026-02-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":110000,"input":100000,"output":8192},"cost":{"input":0.1,"output":0.3,"cache_read":0.01,"cache_write":0.125}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":1024000},"cost":{"input":0.15,"output":0.5,"cache_read":0.15}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-01-28","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":8192},"cost":{"input":0.1,"output":0.3,"cache_read":0.01,"cache_write":0.125}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5-397B-A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-15","last_updated":"2026-05-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":250000,"output":8192},"cost":{"input":0.6,"output":3.6,"cache_read":0.06,"cache_write":0.75}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-10","release_date":"2026-01-10","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"input":40960,"output":0},"cost":{"input":0.01,"output":0}},"NousResearch/Hermes-4-405B":{"id":"NousResearch/Hermes-4-405B","name":"Hermes-4-405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-11","release_date":"2026-01-30","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":120000,"output":8192},"cost":{"input":1,"output":3,"reasoning":3,"cache_read":0.1,"cache_write":1.25}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.3,"output":1.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-01-10","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":124000,"output":8192},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.015,"cache_write":0.18}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8000},"cost":{"input":0.95,"output":4}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8000},"cost":{"input":3,"output":15,"cache_read":3}}}},"minimax-cn-coding-plan":{"id":"minimax-cn-coding-plan","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimaxi.com/anthropic/v1","name":"MiniMax Token Plan (minimaxi.com)","doc":"https://platform.minimaxi.com/docs/token-plan/intro","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"xiaomi-token-plan-ams":{"id":"xiaomi-token-plan-ams","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-ams.xiaomimimo.com/v1","name":"Xiaomi Token Plan (Europe)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}}}},"zeldoc":{"id":"zeldoc","env":["ZELDOC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.zeldoc.ai/v1","name":"Zeldoc","doc":"https://docs.zeldoc.ai","models":{"zdev":{"id":"zdev","name":"ZDev","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}}}},"dinference":{"id":"dinference","env":["DINFERENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.dinference.com/v1","name":"DInference","doc":"https://dinference.com","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.45,"output":1.65}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.25,"output":3.89}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0.22,"output":0.88}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.75,"output":2.4}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.25,"output":3.89}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08","last_updated":"2025-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.0675,"output":0.27}}}},"pioneer":{"id":"pioneer","env":["PIONEER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.pioneer.ai/v1","name":"Pioneer","doc":"https://agent.pioneer.ai/llms.txt","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":0.375}},"devstral-2":{"id":"devstral-2","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.4,"cache_write":0.4}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005,"cache_write":0.05}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":1.5625}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.05,"cache_write":0.1}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.325,"output":1.95,"cache_read":0.065,"cache_write":0.40625}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.2,"cache_write":0.4}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"cache_write":1.5}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.25,"output":1.5,"cache_read":0.03,"cache_write":0.25}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":131072},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":1.25}},"mistral-large-3":{"id":"mistral-large-3","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":1.5,"cache_read":0.5,"cache_write":0.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083333}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25,"cache_write":2.5}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175,"cache_write":1.75}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.15}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":1,"cache_write":2}},"claude-3-7-sonnet-latest":{"id":"claude-3-7-sonnet-latest","name":"Claude Sonnet 3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02,"cache_write":0.2}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_read":0.0375,"cache_write":0.234375}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":240000,"output":64000},"cost":{"input":1.04,"output":6.24,"cache_read":0.208,"cache_write":1.3}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025,"cache_write":0.25}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"mistral-medium-3.5":{"id":"mistral-medium-3.5","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":1.5,"output":7.5,"cache_read":1.5,"cache_write":1.5}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.083333}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"HuggingFaceTB/SmolLM3-3B-Base":{"id":"HuggingFaceTB/SmolLM3-3B-Base","name":"SmolLM3 3B Base","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.27,"output":1.12,"cache_read":0.135,"cache_write":0.27}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.0197,"cache_write":0.1}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":131072},"cost":{"input":0.56,"output":1.68,"cache_read":0.56,"cache_write":0.56}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625,"cache_write":0.435}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01,"cache_write":0.1}},"pioneer/auto":{"id":"pioneer/auto","name":"Pioneer Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2025-06-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":4096}},"mistralai/Mistral-Small-4-119B-2603":{"id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015,"cache_write":0.15}},"mistralai/Pixtral-12B-2409":{"id":"mistralai/Pixtral-12B-2409","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.03,"cache_read":0.02,"cache_write":0.02}},"mistralai/Codestral-22B-v0.1":{"id":"mistralai/Codestral-22B-v0.1","name":"Codestral-22B-v0.1","description":"Open Mistral code model for fill-in-the-middle and 80+ programming languages","family":"codestral","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-05-29","last_updated":"2024-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.3,"output":0.9,"cache_read":0.3,"cache_write":0.3}},"mistralai/Mistral-7B-Instruct-v0.3":{"id":"mistralai/Mistral-7B-Instruct-v0.3","name":"Mistral 7B Instruct v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2023-04-30","last_updated":"2023-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"mistralai/Ministral-8B-Instruct-2410":{"id":"mistralai/Ministral-8B-Instruct-2410","name":"Ministral 8B Instruct","description":"Efficient open Mistral edge model for on-device chat and function calling","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"mistralai/Magistral-Small-2506":{"id":"mistralai/Magistral-Small-2506","name":"Magistral Small","description":"Open Mistral reasoning model for transparent step-by-step problem solving","family":"magistral","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":0.5,"output":1.5,"cache_read":0.5,"cache_write":0.5}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-BF16","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.05,"cache_write":0.05}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":2.5,"cache_read":0.15,"cache_write":0.5}},"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8":{"id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.09,"output":0.45,"cache_read":0.09,"cache_write":0.09}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-E2B-it":{"id":"google/gemma-4-E2B-it","name":"Gemma 4 E2B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-E4B-it":{"id":"google/gemma-4-E4B-it","name":"Gemma 4 E4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"google/diffusiongemma-26B-A4B-it":{"id":"google/diffusiongemma-26B-A4B-it","name":"DiffusionGemma 26B-A4B IT","description":"Gemini model for general assistance, reasoning, and multimodal workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":0.5,"cache_read":0.5,"cache_write":0.5}},"google/gemma-4-12B-it":{"id":"google/gemma-4-12B-it","name":"Gemma 4 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.25,"output":0.25,"cache_read":0.25,"cache_write":0.25}},"google/gemma-3-4b-pt":{"id":"google/gemma-3-4b-pt","name":"Gemma 3 4B (Pretrained)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-02-28","last_updated":"2025-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":131072},"cost":{"input":0.98,"output":3.08,"cache_read":0.182,"cache_write":0.98}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1040000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":1.4}},"LiquidAI/LFM2-24B-A2B":{"id":"LiquidAI/LFM2-24B-A2B","name":"LFM2 24B A2B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-01-31","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.03,"output":0.12,"cache_read":0.03,"cache_write":0.03}},"fastino/gliguard-LLMGuardrails-300M":{"id":"fastino/gliguard-LLMGuardrails-300M","name":"GLiGuard LLM Guardrails 300M","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-base-v1":{"id":"fastino/gliner2-base-v1","name":"GLiNER2 Base","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-multi-v1":{"id":"fastino/gliner2-multi-v1","name":"GLiNER2 Multi","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-30","last_updated":"2025-11-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-multi-large-v1":{"id":"fastino/gliner2-multi-large-v1","name":"GLiNER2 Multi Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-30","last_updated":"2025-11-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-privacy-filter-PII-multi":{"id":"fastino/gliner2-privacy-filter-PII-multi","name":"GLiNER2 Privacy Filter PII (Multi)","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"fastino/gliner2-large-v1":{"id":"fastino/gliner2-large-v1","name":"GLiNER2 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15,"cache_write":1.25}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-03-31","release_date":"2025-03-31","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen3-4B-Instruct-2507":{"id":"Qwen/Qwen3-4B-Instruct-2507","name":"Qwen3 4B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":0.3,"cache_read":0.3,"cache_write":0.3}},"Qwen/Qwen3-1.7B-Base":{"id":"Qwen/Qwen3-1.7B-Base","name":"Qwen3 1.7B Base","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":1.2,"output":1.2,"cache_read":1.2,"cache_write":1.2}},"Qwen/Qwen3-4B-Base":{"id":"Qwen/Qwen3-4B-Base","name":"Qwen3 4B Base","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.15,"output":0.15,"cache_read":0.15,"cache_write":0.15}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.9,"output":0.9,"cache_read":0.9,"cache_write":0.9}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.6,"output":0.6,"cache_read":0.6,"cache_write":0.6}},"Qwen/Qwen2.5-Coder-0.5B":{"id":"Qwen/Qwen2.5-Coder-0.5B","name":"Qwen2.5-Coder-0.5B","description":"Tiny open Qwen code model for lightweight completion and on-device coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":1,"cache_read":0.028,"cache_write":0.175}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.3}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.279,"output":1.2,"cache_read":0.279,"cache_write":0.279}},"meta-llama/Llama-3.2-3B":{"id":"meta-llama/Llama-3.2-3B","name":"Llama-3.2-3B","description":"Small open Llama base model for lightweight text generation and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.2-1B":{"id":"meta-llama/Llama-3.2-1B","name":"Llama-3.2-1B","description":"Compact open Llama base model for lightweight and on-device use","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.1,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-06-30","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.2,"cache_read":0.2,"cache_write":0.2}},"meta-llama/Llama-3.2-3B-Instruct":{"id":"meta-llama/Llama-3.2-3B-Instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-31","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":80000},"cost":{"input":0.1,"output":0.335,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.2-1B-Instruct":{"id":"meta-llama/Llama-3.2-1B-Instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-31","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":60000},"cost":{"input":0.1,"output":0.201,"cache_read":0.1,"cache_write":0.1}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.9,"output":0.9,"cache_read":0.9,"cache_write":0.9}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.07,"output":0.3,"cache_read":0.035,"cache_write":0.07}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.15,"output":0.6,"cache_read":0.015,"cache_write":0.15}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19,"cache_write":0.95}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.34,"cache_write":0.95}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131000},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"cache_write":0.435}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"cache_write":0.14}}}},"helicone":{"id":"helicone","env":["HELICONE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ai-gateway.helicone.ai/v1","name":"Helicone","doc":"https://helicone.ai/models","models":{"llama-3.1-8b-instruct-turbo":{"id":"llama-3.1-8b-instruct-turbo","name":"Meta Llama 3.1 8B Instruct Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.03}},"grok-3-mini":{"id":"grok-3-mini","name":"xAI Grok 3 Mini","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":0.5,"cache_read":0.075}},"gpt-5-nano":{"id":"gpt-5-nano","name":"OpenAI GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.049999999999999996,"output":0.39999999999999997,"cache_read":0.005}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"xAI Grok 4.1 Fast Non-Reasoning","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"llama-3.1-8b-instruct":{"id":"llama-3.1-8b-instruct","name":"Meta Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"cost":{"input":0.02,"output":0.049999999999999996}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"OpenAI GPT-4.1 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.09999999999999999,"output":0.39999999999999997,"cache_read":0.024999999999999998}},"gpt-5-codex":{"id":"gpt-5-codex","name":"OpenAI: GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"claude-3-haiku-20240307":{"id":"claude-3-haiku-20240307","name":"Anthropic: Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-03-07","last_updated":"2024-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"deepseek-v3":{"id":"deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.56,"output":1.68,"cache_read":0.07}},"glm-4.6":{"id":"glm-4.6","name":"Zai GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.44999999999999996,"output":1.5}},"gpt-5-pro":{"id":"gpt-5-pro","name":"OpenAI: GPT-5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":15,"output":120}},"llama-prompt-guard-2-86m":{"id":"llama-prompt-guard-2-86m","name":"Meta Llama Prompt Guard 2 86M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":512,"output":2},"cost":{"input":0.01,"output":0.01}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":16384},"cost":{"input":0.14,"output":1.4}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"OpenAI: GPT-5.1 Codex Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.024999999999999998}},"llama-prompt-guard-2-22m":{"id":"llama-prompt-guard-2-22m","name":"Meta Llama Prompt Guard 2 22M","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":512,"output":2},"cost":{"input":0.01,"output":0.01}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"OpenAI: GPT-5.1 Codex","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"xAI Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-25","last_updated":"2024-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000},"cost":{"input":0.19999999999999998,"output":1.5,"cache_read":0.02}},"kimi-k2-0905":{"id":"kimi-k2-0905","name":"Kimi K2 (09/05)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.5,"output":2,"cache_read":0.39999999999999997}},"gemma2-9b-it":{"id":"gemma2-9b-it","name":"Google Gemma 2","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-25","last_updated":"2024-06-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.01,"output":0.03}},"chatgpt-4o-latest":{"id":"chatgpt-4o-latest","name":"OpenAI ChatGPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-14","last_updated":"2024-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":5,"output":20,"cache_read":2.5}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Google Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09999999999999999,"output":0.39999999999999997,"cache_read":0.024999999999999998,"cache_write":0.09999999999999999}},"ernie-4.5-21b-a3b-thinking":{"id":"ernie-4.5-21b-a3b-thinking","name":"Baidu Ernie 4.5 21B A3B Thinking","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ernie","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-16","last_updated":"2025-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.07,"output":0.28}},"grok-4":{"id":"grok-4","name":"xAI Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-09","last_updated":"2024-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"qwen3-235b-a22b-thinking":{"id":"qwen3-235b-a22b-thinking","name":"Qwen3 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":81920},"cost":{"input":0.3,"output":2.9000000000000004}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":262144},"cost":{"input":0.48,"output":2}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":40960},"cost":{"input":0.29,"output":0.59}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"Google Gemini 3 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.19999999999999998}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Anthropic: Claude Opus 4.1 (20250805)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-4.1-mini-2025-04-14":{"id":"gpt-4.1-mini-2025-04-14","name":"OpenAI GPT-4.1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.39999999999999997,"output":1.5999999999999999,"cache_read":0.09999999999999999}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"OpenAI GPT-4.1 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.39999999999999997,"output":1.5999999999999999,"cache_read":0.09999999999999999}},"sonar-reasoning":{"id":"sonar-reasoning","name":"Perplexity Sonar Reasoning","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":1,"output":5}},"gpt-5-chat-latest":{"id":"gpt-5-chat-latest","name":"OpenAI GPT-5 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-09","release_date":"2024-09-30","last_updated":"2024-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"deepseek-reasoner":{"id":"deepseek-reasoner","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.56,"output":1.68,"cache_read":0.07}},"claude-4.5-opus":{"id":"claude-4.5-opus","name":"Anthropic: Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"OpenAI GPT-OSS 20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.049999999999999996,"output":0.19999999999999998}},"claude-3.5-sonnet-v2":{"id":"claude-3.5-sonnet-v2","name":"Anthropic: Claude 3.5 Sonnet v2","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"qwen3-coder":{"id":"qwen3-coder","name":"Qwen3 Coder 480B A35B Instruct Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.95}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"xAI Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.09999999999999999,"output":0.3}},"gpt-5.1":{"id":"gpt-5.1","name":"OpenAI GPT-5.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"grok-3":{"id":"grok-3","name":"xAI Grok 3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.75}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"OpenAI GPT-5.1 Chat","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-codex","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"o1-mini":{"id":"o1-mini","name":"OpenAI: o1-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":65536},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Meta Llama 4 Maverick 17B 128E","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.6}},"o1":{"id":"o1","name":"OpenAI: o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-4o":{"id":"gpt-4o","name":"OpenAI GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"xAI: Grok 4 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Anthropic: Claude Sonnet 4.5 (20250929)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"mistral-nemo":{"id":"mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16400},"cost":{"input":20,"output":40}},"llama-guard-4":{"id":"llama-guard-4","name":"Meta Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":1024},"cost":{"input":0.21,"output":0.21}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Anthropic: Claude 4.5 Haiku (20251001)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.09999999999999999,"cache_write":1.25}},"deepseek-tng-r1t2-chimera":{"id":"deepseek-tng-r1t2-chimera","name":"DeepSeek TNG R1T2 Chimera","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-02","last_updated":"2025-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":163840},"cost":{"input":0.3,"output":1.2}},"deepseek-r1-distill-llama-70b":{"id":"deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.03,"output":0.13}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"OpenAI GPT-4o-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"claude-3.5-haiku":{"id":"claude-3.5-haiku","name":"Anthropic: Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":0.7999999999999999,"output":4,"cache_read":0.08,"cache_write":1}},"hermes-2-pro-llama-3-8b":{"id":"hermes-2-pro-llama-3-8b","name":"Hermes 2 Pro Llama 3 8B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-05-27","last_updated":"2024-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.14,"output":0.14}},"gpt-4.1":{"id":"gpt-4.1","name":"OpenAI GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"sonar":{"id":"sonar","name":"Perplexity Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":1,"output":1}},"kimi-k2-0711":{"id":"kimi-k2-0711","name":"Kimi K2 (07/11)","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.5700000000000001,"output":2.3}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Perplexity Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":2,"output":8}},"claude-opus-4":{"id":"claude-opus-4","name":"Anthropic: Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":41000,"output":41000},"cost":{"input":0.08,"output":0.29}},"llama-4-scout":{"id":"llama-4-scout","name":"Meta Llama 4 Scout 17B 16E","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.08,"output":0.3}},"deepseek-v3.1-terminus":{"id":"deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.27,"output":1,"cache_read":0.21600000000000003}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Anthropic: Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-3.7-sonnet":{"id":"claude-3.7-sonnet","name":"Anthropic: Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-02","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"mistral-small":{"id":"mistral-small","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.2}},"mistral-large-2411":{"id":"mistral-large-2411","name":"Mistral-Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-24","last_updated":"2024-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":6}},"qwen3-vl-235b-a22b-instruct":{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.3,"output":1.5}},"sonar-pro":{"id":"sonar-pro","name":"Perplexity Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":3,"output":15}},"claude-4.5-sonnet":{"id":"claude-4.5-sonnet","name":"Anthropic: Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"gpt-5-mini":{"id":"gpt-5-mini","name":"OpenAI GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.024999999999999998}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"OpenAI GPT-OSS 120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.04,"output":0.16}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":4096},"cost":{"input":2,"output":8}},"llama-3.1-8b-instant":{"id":"llama-3.1-8b-instant","name":"Meta Llama 3.1 8B Instant","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32678},"cost":{"input":0.049999999999999996,"output":0.08}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Google Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.3125,"cache_write":1.25}},"qwen2.5-coder-7b-fast":{"id":"qwen2.5-coder-7b-fast","name":"Qwen2.5 Coder 7B fast","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-15","last_updated":"2024-09-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.03,"output":0.09}},"claude-4.5-haiku":{"id":"claude-4.5-haiku","name":"Anthropic: Claude 4.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-10","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":1,"output":5,"cache_read":0.09999999999999999,"cache_write":1.25}},"gpt-5":{"id":"gpt-5","name":"OpenAI GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2025-01","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.12500000000000003}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Google Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.3}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.30000000000000004,"cache_write":3.75}},"gemma-3-12b-it":{"id":"gemma-3-12b-it","name":"Google Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.049999999999999996,"output":0.09999999999999999}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Meta Llama 3.3 70B Versatile","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32678},"cost":{"input":0.59,"output":0.7899999999999999}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Meta Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16400},"cost":{"input":0.13,"output":0.39}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"xAI Grok 4 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.19999999999999998,"output":0.5,"cache_read":0.049999999999999996}},"o4-mini":{"id":"o4-mini","name":"OpenAI o4 Mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"o3-mini":{"id":"o3-mini","name":"OpenAI o3 Mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2023-10","release_date":"2023-10-01","last_updated":"2023-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o3":{"id":"o3","name":"OpenAI o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-pro":{"id":"o3-pro","name":"OpenAI o3 Pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":false,"reasoning":false,"tool_call":true,"temperature":false,"knowledge":"2024-06","release_date":"2024-06-01","last_updated":"2024-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}}}},"cloudferro-sherlock":{"id":"cloudferro-sherlock","env":["CLOUDFERRO_SHERLOCK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-sherlock.cloudferro.com/openai/v1/","name":"CloudFerro Sherlock","doc":"https://docs.sherlock.cloudferro.com/","models":{"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"input":180000,"output":16000},"cost":{"input":0.3,"output":1.2}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10-09","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":70000,"output":70000},"cost":{"input":2.92,"output":2.92}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":2.92,"output":2.92}},"speakleash/Bielik-11B-v2.6-Instruct":{"id":"speakleash/Bielik-11B-v2.6-Instruct","name":"Bielik 11B v2.6 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.67,"output":0.67}},"speakleash/Bielik-11B-v3.0-Instruct":{"id":"speakleash/Bielik-11B-v3.0-Instruct","name":"Bielik 11B v3.0 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.67,"output":0.67}}}},"stepfun":{"id":"stepfun","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.com/v1","name":"StepFun (China)","doc":"https://platform.stepfun.com/docs/zh/overview/concept","models":{"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-06-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.185,"output":1.11,"cache_read":0.037}},"step-tts-2":{"id":"step-tts-2","name":"Step TTS 2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"step-2-16k":{"id":"step-2-16k","name":"Step 2 (16K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"input":16384,"output":8192},"cost":{"input":5.21,"output":16.44,"cache_read":1.04}},"stepaudio-2.5-asr":{"id":"stepaudio-2.5-asr","name":"StepAudio 2.5 ASR","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-24","last_updated":"2026-07-02","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"stepaudio-2.5-tts":{"id":"stepaudio-2.5-tts","name":"StepAudio 2.5 TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"step","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"step-1-32k":{"id":"step-1-32k","name":"Step 1 (32K)","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-01-01","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":2.05,"output":9.59,"cache_read":0.41}}}},"unorouter":{"id":"unorouter","env":["UNOROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.unorouter.com/v1","name":"UnoRouter","doc":"https://unorouter.com/models","models":{"deepseek-v4-pro:free":{"id":"deepseek-v4-pro:free","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.819,"output":3.276}},"deepseek-v4-flash:free":{"id":"deepseek-v4-flash:free","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.2675,"output":5.3368}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.6001,"output":5.0288}},"qwen3.5-397b-a17b:free":{"id":"qwen3.5-397b-a17b:free","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.0625,"output":0.125}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1.8,"output":10.8}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1857,"output":1.1142}},"glm-4.5-flash:free":{"id":"glm-4.5-flash:free","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.2,"output":6}},"step-3.7-flash:free":{"id":"step-3.7-flash:free","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"nemotron-3-ultra-550b-a55b:free":{"id":"nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0}},"gemma-4-31b-it:free":{"id":"gemma-4-31b-it:free","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.2:free":{"id":"glm-5.2:free","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"minimax-m2.7:free":{"id":"minimax-m2.7:free","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"gpt-5.4:free":{"id":"gpt-5.4:free","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"gpt-5.5:free":{"id":"gpt-5.5:free","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.425,"output":2.125}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.8999,"output":1.7999}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.05,"output":8.4}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.44,"output":7.2}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.1875,"output":1.125}}}},"coralbricks":{"id":"coralbricks","env":["CORAL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.coralbricks.ai/v1","name":"CoralBricks","doc":"https://www.coralbricks.ai/docs","models":{"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0}},"glm-5.3-fp4":{"id":"glm-5.3-fp4","name":"GLM 5.3 FP4","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.12,"output":4.4,"cache_read":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.12,"output":0.6,"cache_read":0}}}},"hyper":{"id":"hyper","env":["HYPER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://hyper.charm.land/v1","name":"Charm Hyper","doc":"https://hyper.charm.land","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":25600},"cost":{"input":0.122,"output":0.42,"cache_read":0.061}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.437216,"output":4.311648,"cache_read":0.047907}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.044}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":26214},"cost":{"input":0.1175,"output":1.136,"cache_read":0.05875}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-20","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-05","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":6553},"cost":{"input":0.404,"output":1.496,"cache_read":0.202}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-07-03","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":26214},"cost":{"input":1.03436,"output":4.3552,"cache_read":0.174208}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":430000,"output":43000},"cost":{"input":0.274,"output":0.8992,"cache_read":0.137}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.52432,"output":4.79072,"cache_read":0.152432}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.32664,"output":1.30656,"cache_read":0.064239}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-06","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-07-03","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":16000},"cost":{"input":1.03436,"output":4.3552,"cache_read":0.206872}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":26214},"cost":{"input":0.6,"output":2.5,"cache_read":0.3}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-27","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":64000},"cost":{"input":0.2,"output":0.8,"cache_read":0.04}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16000},"cost":{"input":3.2664,"output":16.332,"cache_read":0.32664}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16332,"output":0.5444,"cache_read":0.031575}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-20","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1,"output":4,"cache_read":0.1,"cache_write":1.25}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-15","last_updated":"2026-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1.0888,"output":4.40964,"cache_read":0.185096}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-04-13","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":20275},"cost":{"input":0.86,"output":2.784,"cache_read":0.43}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-13","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":26214},"cost":{"input":0.5584,"output":2.935,"cache_read":0.2792}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202750,"output":3276},"cost":{"input":1.326,"output":4.22,"cache_read":0.663}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-15","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.2,"output":4.8,"cache_read":0.24}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-06","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":2.4,"output":4.8,"cache_read":0.2}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-13","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128072,"output":13107},"cost":{"input":0.178,"output":0.68,"cache_read":0.089}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.52432,"output":4.79072,"cache_read":0.283088}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":12800},"cost":{"input":0.6066,"output":1.0386,"cache_read":0.3033}},"qwen3.6-max":{"id":"qwen3.6-max","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-05-20","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"qwen3-coder-480b-a35b-instruct-int4-mixed-ar":{"id":"qwen3-coder-480b-a35b-instruct-int4-mixed-ar","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-30","last_updated":"2026-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":106000,"output":10600},"cost":{"input":0.445,"output":2.145,"cache_read":0.2225}}}},"requesty":{"id":"requesty","env":["REQUESTY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://router.requesty.ai/v1","name":"Requesty","doc":"https://requesty.ai/solution/llm-routing/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7@eu":{"id":"claude-opus-4-7@eu","name":"Claude Opus 4.7 (EU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.25,"cache_write":3.125}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.34,"cache_read":0.07}},"glm-5.3@eu":{"id":"glm-5.3@eu","name":"GLM-5.3 (EU)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"kimi-k2.7-code@eu":{"id":"kimi-k2.7-code@eu","name":"Kimi K2.7 Code (EU)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.25,"output":4.5,"cache_read":0.31}},"qwen3.8-2.4T-A95B":{"id":"qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"nemotron-3.5-content-safety":{"id":"nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0,"output":0}},"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"mistral-medium-3-5":{"id":"mistral-medium-3-5","name":"mistral-medium-3-5","description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":1.65,"output":8.25,"cache_read":1.65}},"ring-2.6-1t":{"id":"ring-2.6-1t","name":"ring-2.6-1t","description":"Inclusion AI ring-2.6-1t","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-05-08","last_updated":"2026-05-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.5}},"gpt-4.1-mini@eu":{"id":"gpt-4.1-mini@eu","name":"GPT-4.1 mini (EU)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.44,"output":1.76,"cache_read":0.11}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"gemini-3.7-flash@eu":{"id":"gemini-3.7-flash@eu","name":"Gemini 3.7 Flash (EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"qwen3.8-flash-next":{"id":"qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"leanstral-1-5@eu":{"id":"leanstral-1-5@eu","name":"leanstral-1-5@eu","description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813@eu":{"id":"deepseek-v4-pro-0813@eu","name":"DeepSeek V4 Pro 0813 (EU)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.44}},"ling-3.0-tiny":{"id":"ling-3.0-tiny","name":"ling-3.0-tiny","description":"Ling-3.0-tiny is an efficient 7.9B parameter MoE model from inclusionAI with only 1.3B active parameters per token. Built for responsive agents, reliable instruction following and multi turn conversation, with a 256K context window, native function calling, prompt caching and switchable Thinking and Instant modes.","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":1.25,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5}}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"nvidia-nemotron-3-ultra":{"id":"nvidia-nemotron-3-ultra","name":"nvidia-nemotron-3-ultra","description":"NVIDIA Nemotron 3 Ultra is NVIDIA's strongest open-weights reasoning model, positioned near GPT-5.4 Mini (xhigh) and ahead of DeepSeek V4-Flash and Qwen3.5-397B-A17B.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":2.5}},"nvidia-nemotron-3-super-120b-a12b":{"id":"nvidia-nemotron-3-super-120b-a12b","name":"nvidia-nemotron-3-super-120b-a12b","description":"NVIDIA Nemotron 3 Super is a hybrid Mixture-of-Experts (MoE) model engineered for highest compute efficiency and accuracy in multi-agent applications and specialized agentic systems. It is optimized to run many collaborating agents per application on a single GPU, delivering high accuracy for reasoning, tool use, and instruction following.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.5}},"minimax-m2.7-highspeed":{"id":"minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":1.2}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"claude-fable-5.1@eu":{"id":"claude-fable-5.1@eu","name":"Claude Fable 5.1 (EU)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.26,"output":2.6}},"claude-opus-4-6@eu":{"id":"claude-opus-4-6@eu","name":"Claude Opus 4.6 (EU)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":1,"cache_read":0.05}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":1.2}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":9,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":9}}},"kat-coder-pro":{"id":"kat-coder-pro","name":"kat-coder-pro","description":"KAT-Coder-Pro V2 by KwaiKAT is a non-reasoning model optimized for agentic coding. It delivers strong performance on reasoning-style tasks while requiring significantly fewer output tokens than peer models. With the 1210 release, it achieved a score of 64 on the Artificial Analysis Intelligence Index, placing it in the global Top 10 and ranking first among all non-reasoning models.","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":1.2}},"gpt-5.6-terra@eu":{"id":"gpt-5.6-terra@eu","name":"GPT-5.6 Terra (EU)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"gpt-5.4@eu":{"id":"gpt-5.4@eu","name":"GPT-5.4 (EU)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"tiers":[{"input":20,"output":75,"cache_read":2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2}}},"gemini-3.8-flash@eu":{"id":"gemini-3.8-flash@eu","name":"Gemini 3.8 Flash (EU)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.0825}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-nano@eu":{"id":"gpt-5-nano@eu","name":"GPT-5 Nano (EU)","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.055,"output":0.44,"cache_read":0.0055}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"claude-sonnet-5@eu":{"id":"claude-sonnet-5@eu","name":"Claude Sonnet 5 (EU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.5,"output":7,"cache_read":0.15}},"nemotron-3-nano-omni-30b-a3b-reasoning":{"id":"nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":20480},"cost":{"input":0,"output":0}},"gpt-5.5@eu":{"id":"gpt-5.5@eu","name":"GPT-5.5 (EU)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.75,"output":16.5,"cache_read":0.275}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"seed-1.8":{"id":"seed-1.8","name":"seed-1.8","description":"Optimized specifically for multimodal agent scenarios. It features enhanced agent capabilities, upgraded multimodal comprehension, and more flexible context management.","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.1}},"gpt-5-mini@eu":{"id":"gpt-5-mini@eu","name":"GPT-5 Mini (EU)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":0.275,"output":2.2,"cache_read":0.0275}},"gpt-4.1-nano@eu":{"id":"gpt-4.1-nano@eu","name":"GPT-4.1 nano (EU)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.11,"output":0.44,"cache_read":0.0275}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"claude-fable-5@eu":{"id":"claude-fable-5@eu","name":"Claude Fable 5 (EU)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"ling-2.6-1t":{"id":"ling-2.6-1t","name":"ling-2.6-1t","description":"Inclusion AI ling-2.6-1t","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.5}},"mistral-medium-latest":{"id":"mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"seed-2.0-pro":{"id":"seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"glm-5.1@eu":{"id":"glm-5.1@eu","name":"GLM-5.1 (EU)","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"qwen3.8-flash-next@eu":{"id":"qwen3.8-flash-next@eu","name":"Qwen3.8 Flash Next (EU)","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"thinkingcap-qwen3.6-27b":{"id":"thinkingcap-qwen3.6-27b","name":"thinkingcap-qwen3.6-27b","description":"ThinkingCap-Qwen3.6-27B is a reasoning tuned model from BottlecapAI built on Qwen3.6 27B. It supports extended thinking with tool calling and a 256K context window. Served via Sference.","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-13","last_updated":"2026-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":2.6,"cache_read":0.05}},"nemotron-3-ultra-nvfp4":{"id":"nemotron-3-ultra-nvfp4","name":"nemotron-3-ultra-nvfp4","description":"Nemotron-3-Ultra-550B-A55B-NVFP4 is a frontier-scale large language model (LLM) trained by NVIDIA, designed to deliver strong agentic, reasoning, and conversational capabilities. It is optimized for the most demanding workloads, including complex multi-step agents, long-context analysis, and high-accuracy reasoning over code, math, and science. The model employs a hybrid Latent Mixture-of-Experts (LatentMoE) architecture, utilizing interleaved Mamba-2 and MoE layers, along with select Attention layers. Like the Super model, the Ultra model incorporates Multi-Token Prediction (MTP) layers for faster text generation and improved quality, and it is trained using an NVFP4 pre-training recipe to maximize compute efficiency. The model has 55B active parameters and 550B parameters in total.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax-m3@eu":{"id":"minimax-m3@eu","name":"MiniMax-M3 (EU)","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.4,"output":2,"cache_read":0.1}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"devstral-latest@eu":{"id":"devstral-latest@eu","name":"devstral-latest@eu","description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":1.583}},"mistral-medium-3-5@eu":{"id":"mistral-medium-3-5@eu","name":"mistral-medium-3-5@eu","description":"Mistral Medium 3.5 is a dense 128B instruction following model from Mistral AI. It supports text and image inputs with text output, and is designed for agentic workflows, coding, and complex multi step reasoning. It is particularly strong at reliable multi tool calling and long horizon tasks, with a 256K context window, configurable reasoning effort per request, and a custom vision encoder that handles variable image sizes and aspect ratios. Self hostable on as few as four GPUs and available under open weights.","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":1.65,"output":8.25,"cache_read":1.65}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04}}},"nemotron-lightning-3.5-30b-a3b":{"id":"nemotron-lightning-3.5-30b-a3b","name":"nemotron-lightning-3.5-30b-a3b","description":"Nemotron-Lightning-3.5-30B-A3B is a 30B-parameter Mixture-of-Experts language model (3B active) from NVIDIA's Nemotron-H family, built on a hybrid Mamba-Transformer architecture for efficient long-context inference. Like other models in the family, it responds to queries by first generating a reasoning trace and then concluding with a final response, with reasoning behavior configurable through a flag in the chat template. It includes a multi-token prediction (MTP) speculative decoding head for low-latency serving.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-15","last_updated":"2026-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"seed-2.0-code":{"id":"seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15,"cache_read":0.45}},"mistral-medium-latest@eu":{"id":"mistral-medium-latest@eu","name":"Mistral Medium (latest) (EU)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"gpt-5.6-sol@eu":{"id":"gpt-5.6-sol@eu","name":"GPT-5.6 Sol (EU)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5.5,"output":33,"cache_read":0.55}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"glm-5.2-fast","description":"GLM-5.2 introduces a robust 1M-token context and advanced, multi-effort coding capabilities to significantly enhance performance on long-horizon tasks. Its new IndexShare architecture and improved MTP layer simultaneously boost efficiency by reducing per-token FLOPs and increasing speculative decoding lengths. A 743B-parameter model in Zhipu AI's GLM series, designed to plan, execute, and iterate autonomously on extended, engineering-grade tasks.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-13","last_updated":"2026-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":2}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0.2,"output":0.6,"cache_read":0.07}},"claude-sonnet-4-6@eu":{"id":"claude-sonnet-4-6@eu","name":"Claude Sonnet 4.6 (EU)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.3,"cache_write":4.125}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"laguna-m.1":{"id":"laguna-m.1","name":"Laguna M.1","description":"Poolside's open-weight model for agentic coding and long-horizon work","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0,"output":0}},"leanstral-1-5":{"id":"leanstral-1-5","name":"leanstral-1-5","description":"Leanstral 1.5 is an updated Lean 4 formal proof engineering model from Mistral AI, optimized for automated theorem proving and autoformalization. It has 119B total parameters with 6.5B active and supports a 256K token context window. It supports native function calling and structured output.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"glm-5.3-flash@eu":{"id":"glm-5.3-flash@eu","name":"GLM-5.3-Flash (EU)","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0.2,"output":0.6,"cache_read":0.07}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"laguna-xs.2":{"id":"laguna-xs.2","name":"Laguna XS.2","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0,"output":0}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"nemotron-3-nano-omni":{"id":"nemotron-3-nano-omni","name":"nemotron-3-nano-omni","description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":300000},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"nemotron-3-super-120b-a12b":{"id":"nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gpt-5@eu":{"id":"gpt-5@eu","name":"GPT-5 (EU)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":11,"cache_read":0.1375}},"thinkingcap-qwen3.6-27b@eu":{"id":"thinkingcap-qwen3.6-27b@eu","name":"thinkingcap-qwen3.6-27b@eu","description":"ThinkingCap-Qwen3.6-27B is a reasoning tuned model from BottlecapAI built on Qwen3.6 27B. It supports extended thinking with tool calling and a 256K context window. Served via Sference.","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-13","last_updated":"2026-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":2.6,"cache_read":0.05}},"seed-2.0-mini":{"id":"seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02}},"kimi-k2.6@eu":{"id":"kimi-k2.6@eu","name":"Kimi K2.6 (EU)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":128000},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":20480},"cost":{"input":0,"output":0}},"nemotron-3-nano-omni@eu":{"id":"nemotron-3-nano-omni@eu","name":"nemotron-3-nano-omni@eu","description":"The most open, efficient, and accurate omni modal reasoning model for agentic AI.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-05-20","last_updated":"2026-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":300000},"cost":{"input":0.06,"output":0.24,"cache_read":0.06}},"ling-2.6-flash":{"id":"ling-2.6-flash","name":"ling-2.6-flash","description":"Inclusion AI ling-2.6-flash","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3}},"deepseek-v4-pro@eu":{"id":"deepseek-v4-pro@eu","name":"DeepSeek V4 Pro (EU)","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.75,"output":3.5,"cache_read":0.44}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.16,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"claude-sonnet-4@eu":{"id":"claude-sonnet-4@eu","name":"Claude Sonnet 4 (latest) (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"gemini-3.5-flash-lite@eu":{"id":"gemini-3.5-flash-lite@eu","name":"Gemini 3.5 Flash Lite (EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.33,"output":2.75,"cache_read":0.033}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"claude-opus-5@eu":{"id":"claude-opus-5@eu","name":"Claude Opus 5 (EU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"glm-5.2@eu":{"id":"glm-5.2@eu","name":"GLM-5.2 (EU)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"claude-haiku-4-5@eu":{"id":"claude-haiku-4-5@eu","name":"Claude Haiku 4.5 (latest) (EU)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"gpt-4o-mini@eu":{"id":"gpt-4o-mini@eu","name":"GPT-4o mini (EU)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.165,"output":0.66,"cache_read":0.0825}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"devstral-latest":{"id":"devstral-latest","name":"devstral-latest","description":"An enterprise grade text model, that excels at using tools to explore codebases, editing multiple files and power software engineering agents.","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-05-27","last_updated":"2026-05-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.44,"output":2.2,"cache_read":0.44}},"kimi-k3@eu":{"id":"kimi-k3@eu","name":"Kimi K3 (EU)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15,"cache_read":0.45}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.032,"cache_write":0.4}},"gemini-2.5-flash-lite@eu":{"id":"gemini-2.5-flash-lite@eu","name":"Gemini 2.5 Flash-Lite (EU)","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.18333}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.165,"output":0.66,"cache_read":0.165}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4.2-beta":{"id":"grok-4.2-beta","name":"grok-4.2-beta","description":"Grok 4.20 Beta is xAI's newest flagship model with industry-leading speed and agentic tool calling capabilities. It combines the lowest hallucination rate on the market with strict prompt adherance, delivering consistently precise and truthful responses.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":2,"output":6,"cache_read":0.2,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":0.4,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.4,"cache_write":4}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"claude-sonnet-4-5@eu":{"id":"claude-sonnet-4-5@eu","name":"Claude Sonnet 4.5 (latest) (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.3,"cache_write":4.125,"tiers":[{"input":6.6,"output":24.75,"cache_read":0.6,"cache_write":8.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6.6,"output":24.75,"cache_read":0.6,"cache_write":8.25}}},"mistral-small-2603@eu":{"id":"mistral-small-2603@eu","name":"Mistral Small 4 (EU)","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.165,"output":0.66,"cache_read":0.165}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"qwen3.5-2b","description":"Qwen3.5-2B is a compact yet capable model from Alibaba's Qwen3.5 series. It features a 262K token context window, support for 201 languages, thinking/reasoning mode, and tool calling for agentic workflows. A strong choice for prototyping, fine-tuning, and efficient multilingual deployments.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.02,"output":0.1}},"claude-opus-4-5@eu":{"id":"claude-opus-4-5@eu","name":"Claude Opus 4.5 (latest) (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":30}},"gemini-3.1-flash-lite@eu":{"id":"gemini-3.1-flash-lite@eu","name":"Gemini 3.1 Flash Lite (EU)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.275,"output":1.65,"cache_read":0.0275,"cache_write":0.091663}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gemini-2.5-flash@eu":{"id":"gemini-2.5-flash@eu","name":"Gemini 2.5 Flash (EU)","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"cache_read":0.075,"cache_write":0.55}},"gemini-2.5-pro@eu":{"id":"gemini-2.5-pro@eu","name":"Gemini 2.5 Pro (EU)","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":10,"cache_read":0.31,"cache_write":2.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.62,"cache_write":4.75,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.62,"cache_write":4.75}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"nemotron-3-ultra-550b-a55b":{"id":"nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"nemotron-3.5-lightning-30b-a3b":{"id":"nemotron-3.5-lightning-30b-a3b","name":"nemotron-3.5-lightning-30b-a3b","description":"NVIDIA Nemotron 3.5 Lightning 30B-A3B is a hybrid Mamba-2 + MoE + Attention model with 30B total and 3B active parameters, pre-trained on over 20T tokens with an NVFP4 recipe and Multi-Token Prediction for fast generation. Up to 1M token context for long-running autonomous agents, sub-agent workhorse deployments, and agentic workflows. Supports reasoning and tool calling. English and coding languages plus Spanish, French, German, Italian, and Japanese. Open weights under the OpenMDW License Agreement v1.1. Part of the NVIDIA Nemotron family.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"inkling-256k":{"id":"inkling-256k","name":"inkling-256k","description":"Inkling 256K is the extended context variant of Inkling, a large MoE hybrid reasoning model from Thinking Machines with audio and vision input support and a 256K context window.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"gemini-3.5-flash@eu":{"id":"gemini-3.5-flash@eu","name":"Gemini 3.5 Flash (EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.65,"output":9.9,"cache_read":0.165,"cache_write":1.7413}},"gpt-5.6-luna@eu":{"id":"gpt-5.6-luna@eu","name":"GPT-5.6 Luna (EU)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022}},"claude-opus-4-8@eu":{"id":"claude-opus-4-8@eu","name":"Claude Opus 4.8 (EU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"o4-mini@eu":{"id":"o4-mini@eu","name":"o4-mini (EU)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.21,"output":4.84,"cache_read":0.3025}},"gpt-4.1@eu":{"id":"gpt-4.1@eu","name":"GPT-4.1 (EU)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2.2,"output":8.8,"cache_read":0.55}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"deepseek-v4-flash-0731@eu":{"id":"deepseek-v4-flash-0731@eu","name":"DeepSeek V4 Flash 0731 (EU)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"claude-fable-5.1":{"id":"claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"qwen3.8-2.4T-A95B@eu":{"id":"qwen3.8-2.4T-A95B@eu","name":"Qwen3.8 2.4T A95B (EU)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":2.5,"output":6,"cache_read":0.63}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"gpt-5.1@eu":{"id":"gpt-5.1@eu","name":"GPT-5.1 (EU)","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":11,"cache_read":0.1375}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5.5,"output":33,"cache_read":0.55}}}},"llmtr":{"id":"llmtr","env":["LLMTR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llmtr.com/v1","name":"LLMTR","doc":"https://llmtr.com/docs","models":{"medgemma-4b":{"id":"medgemma-4b","name":"MedGemma 4B","description":"Multimodal medical-domain Gemma variant for text and image analysis","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-08-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"status":"deprecated","cost":{"input":3,"output":5}},"muse-glimmer-30b-tr":{"id":"muse-glimmer-30b-tr","name":"Muse Glimmer 30B (TR)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":2,"output":5,"cache_read":0.5}},"gemma-4":{"id":"gemma-4","name":"Gemma 4","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":2,"output":5,"cache_read":0.5}},"magibu-11b-v8":{"id":"magibu-11b-v8","name":"Magibu 11B v8","description":"Turkish-language chat model for instruction following and assistant flows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-05","last_updated":"2026-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":0.1,"output":0.5}},"qwen3-6-35b":{"id":"qwen3-6-35b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":5,"output":10}},"trendyol-asure-12b":{"id":"trendyol-asure-12b","name":"Trendyol Asure 12B","description":"Turkish-language multimodal instruct model built on Gemma 3 12B for e-commerce text, chat, and image-text tasks","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-19","last_updated":"2026-02-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0.1,"output":0.5,"cache_read":0.025}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4}},"qwen/qwen3-vl-plus":{"id":"qwen/qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"cost":{"input":0.2,"output":1.6}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.2}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mimo/mimo-v2.5":{"id":"mimo/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.28}},"mimo/mimo-v2.5-pro":{"id":"mimo/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.435,"output":0.87}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.1}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.58,"output":1.44}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.87,"output":4.68}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2}},"publicai/apertus-8b-instruct":{"id":"publicai/apertus-8b-instruct","name":"Apertus 8B Instruct","description":"Fully open 8B multilingual LLM supporting 1800+ languages with 65K context. Trained on compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0.1,"output":0.2}},"publicai/apertus-70b-instruct":{"id":"publicai/apertus-70b-instruct","name":"Apertus 70B Instruct","description":"Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0.82,"output":2.92}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.03,"output":0.12}},"upstage/solar-pro3":{"id":"upstage/solar-pro3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro2":{"id":"upstage/solar-pro2","name":"Solar Pro 2","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.15,"output":0.6}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}}}},"xiaomi":{"id":"xiaomi","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.xiaomimimo.com/v1","name":"Xiaomi","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-pro-ultraspeed":{"id":"mimo-v2.5-pro-ultraspeed","name":"MiMo-V2.5-Pro-UltraSpeed","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-06-08","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"status":"beta","cost":{"input":1.305,"output":2.61,"cache_read":0.0108}},"mimo-v2-flash":{"id":"mimo-v2-flash","name":"MiMo-V2-Flash","description":"Legacy model retained for compatibility with older integrations","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"mimo-v2-omni":{"id":"mimo-v2-omni","name":"MiMo-V2-Omni","description":"Legacy model retained for compatibility with older integrations","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-06-24","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"status":"deprecated","cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-06-24","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-06-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}}}},"huggingface":{"id":"huggingface","env":["HF_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://router.huggingface.co/v1","name":"Hugging Face","doc":"https://huggingface.co/docs/inference-providers","models":{"stepfun-ai/Step-3.7-Flash":{"id":"stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.1,"output":0.3}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":8192},"cost":{"input":0.4,"output":1.3}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.27,"output":1.12}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.44,"output":1.32}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":3,"output":5}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":32768},"cost":{"input":0.7,"output":2.5}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.28,"output":0.4}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.4}},"zai-org/GLM-4.6V-Flash":{"id":"zai-org/GLM-4.6V-Flash","name":"GLM-4.6V-Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/GLM-4.5V":{"id":"zai-org/GLM-4.5V","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8}},"zai-org/GLM-4.5":{"id":"zai-org/GLM-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":1.4,"output":4.4}},"zai-org/GLM-4.7-Flash":{"id":"zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0,"output":0}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.5,"output":1.2}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.07,"output":0.26}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.17,"output":0.25}},"Qwen/Qwen3-Coder-Next":{"id":"Qwen/Qwen3-Coder-Next","name":"Qwen3-Coder-Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-30B-A3B":{"id":"Qwen/Qwen3-30B-A3B","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"Qwen/Qwen3-235B-A22B":{"id":"Qwen/Qwen3-235B-A22B","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2.5,"output":6.25}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.855,"output":2.565}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":1.5}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":2,"output":2}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.25,"output":1}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3.6}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.29,"output":0.59}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":3}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.47,"output":3.19}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.95}},"Qwen/Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"Qwen/Qwen3-Embedding-8B":{"id":"Qwen/Qwen3-Embedding-8B","name":"Qwen 3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.01,"output":0}},"Qwen/Qwen3-Next-80B-A3B-Thinking":{"id":"Qwen/Qwen3-Next-80B-A3B-Thinking","name":"Qwen3-Next-80B-A3B-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-11","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.3,"output":2}},"Qwen/Qwen3-Embedding-4B":{"id":"Qwen/Qwen3-Embedding-4B","name":"Qwen 3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2048},"cost":{"input":0.01,"output":0}},"Qwen/Qwen2.5-Coder-32B-Instruct":{"id":"Qwen/Qwen2.5-Coder-32B-Instruct","name":"Qwen2.5-Coder-32B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.06,"output":0.2}},"MiniMaxAI/MiniMax-M2":{"id":"MiniMaxAI/MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.1":{"id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-10","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.06,"output":0.06}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.59,"output":0.79}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.5}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.25,"output":0.69}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi-K2-Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi-K2-Instruct-0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":1,"output":3}},"moonshotai/Kimi-K2-Instruct":{"id":"moonshotai/Kimi-K2-Instruct","name":"Kimi-K2-Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-07-14","last_updated":"2025-07-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":3}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15}},"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"MiMo model for long-context reasoning, perception, and agentic tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.4,"output":2}},"XiaomiMiMo/MiMo-V2-Flash":{"id":"XiaomiMiMo/MiMo-V2-Flash","name":"MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.1,"output":0.3}}}},"zhipuai-coding-plan":{"id":"zhipuai-coding-plan","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/coding/paas/v4","name":"Zhipu AI Coding Plan","doc":"https://docs.bigmodel.cn/cn/coding-plan/overview","models":{"glm-5.3-highspeed":{"id":"glm-5.3-highspeed","name":"GLM-5.3 Highspeed","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.2-highspeed":{"id":"glm-5.2-highspeed","name":"GLM-5.2 Highspeed","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"daoxe":{"id":"daoxe","env":["DAOXE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://daoxe.com/v1","name":"DaoXE","doc":"https://daoxe.com/pricing","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}}}},"crossmodel":{"id":"crossmodel","env":["CROSSMODEL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.crossmodel.ai/v1","name":"CrossModel","doc":"https://www.crossmodel.ai/docs","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.88,"output":5.63,"cache_read":0.375,"cache_write":2.35}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.32,"output":1.88,"cache_read":0.032,"cache_write":0.4,"tiers":[{"input":1.25,"output":7.5,"cache_read":0.124,"cache_write":1.57,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.25,"output":7.5,"cache_read":0.124,"cache_write":1.57}}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.04,"output":0.13,"cache_read":0.01,"cache_write":0.04,"tiers":[{"input":0.1,"output":0.37,"cache_read":0.02,"cache_write":0.12,"tier":{"type":"context","size":32000}},{"input":0.19,"output":0.74,"cache_read":0.04,"cache_write":0.24,"tier":{"type":"context","size":256000}}]}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.19,"output":1.13,"cache_read":0.019,"cache_write":0.24,"tiers":[{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.94,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.94}}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.13,"output":0.43,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.88,"output":5.63,"cache_read":0.23,"cache_write":2.35}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.25,"cache_read":0.032,"cache_write":0.4,"tiers":[{"input":0.96,"output":3.75,"cache_read":0.096,"cache_write":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.96,"output":3.75,"cache_read":0.096,"cache_write":1.2}}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.16,"output":0.32,"cache_read":0.004,"cache_write":0.16}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.47,"output":0.94,"cache_read":0.005,"cache_write":0.47}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.33,"output":1.32,"cache_read":0.066,"cache_write":0.42}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":512000},"cost":{"input":0.33,"output":1.32,"cache_read":0.066,"cache_write":0.33,"tiers":[{"input":0.66,"output":2.63,"cache_read":0.132,"cache_write":0.66,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.66,"output":2.63,"cache_read":0.132,"cache_write":0.66}}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":1,"output":4.16,"cache_read":0.18,"cache_write":1}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":1,"output":4.16,"cache_read":0.18,"cache_write":1}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.27,"output":1.08,"cache_read":0.0054,"cache_write":0.27}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.215,"output":3.645,"cache_read":0.0405,"cache_write":1.215}},"gemini/gemini-3.1-pro-preview":{"id":"gemini/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":4}}},"gemini/gemini-2.5-flash-lite":{"id":"gemini/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.1}},"gemini/gemini-3.6-flash":{"id":"gemini/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-3.5-flash":{"id":"gemini/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":1.5}},"gemini/gemini-3.5-flash-lite":{"id":"gemini/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"gemini/gemini-3-flash-preview":{"id":"gemini/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.5}},"gemini/gemini-3.8-flash":{"id":"gemini/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-3.7-flash":{"id":"gemini/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.75}},"gemini/gemini-2.5-pro":{"id":"gemini/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":1.25,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5}}},"gemini/gemini-2.5-flash":{"id":"gemini/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.3}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":1.25,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":2.5}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":0.6,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6,"cache_write":4}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"cache_write":1,"tiers":[{"input":2,"output":4,"cache_read":0.4,"cache_write":2,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4,"cache_write":2}}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"cache_write":2,"tiers":[{"input":4,"output":12,"cache_read":1,"cache_write":4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1,"cache_write":4}}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":2.5,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":5}}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.15}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02,"cache_write":0.2}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075,"cache_write":0.75}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":10}}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":131072},"cost":{"input":0.16,"output":0.64,"cache_read":0.04,"cache_write":0.16}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":0.96,"output":2.88,"cache_read":0.048,"cache_write":0.96}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.47,"output":2.16,"cache_read":0.1,"cache_write":0.47,"tiers":[{"input":0.62,"output":2.47,"cache_read":0.13,"cache_write":0.62,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03,"cache_write":0.15}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":3,"cache_read":0.16,"cache_write":0.6,"tiers":[{"input":0.8,"output":3.4,"cache_read":0.2,"cache_write":0.8,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1,"output":3.8,"cache_read":0.2,"cache_write":1,"tiers":[{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.9,"output":3.7,"cache_read":0.18,"cache_write":0.9,"tiers":[{"input":1.1,"output":4.3,"cache_read":0.27,"cache_write":1.1,"tier":{"type":"context","size":32000}}]}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":1.2,"output":4.4,"cache_read":0.3,"cache_write":1.2}}}},"minimax":{"id":"minimax","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.io/anthropic/v1","name":"MiniMax (minimax.io)","doc":"https://platform.minimax.io/docs/guides/quickstart","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}}}},"salad-cloud":{"id":"salad-cloud","env":["SALAD_CLOUD_API_KEY"],"npm":"@saladtechnologies-oss/ai-sdk-provider","name":"SaladCloud AI Gateway","doc":"https://docs.salad.com/ai-gateway/explanation/overview","models":{"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Qwen MoE for agentic tasks, complex reasoning, code generation, and instruction following","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":262144,"output":262144},"cost":{"input":0.09,"output":0.6}}}},"aki-io":{"id":"aki-io","env":["AKI_IO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://aki.io/v1","name":"AKI.IO","doc":"https://aki.io/docs/","models":{"gemma4-26b":{"id":"gemma4-26b","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.5}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.3,"output":2.2,"cache_read":0.1}},"deepseek-v4-flash-0731-284b":{"id":"deepseek-v4-flash-0731-284b","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":81920},"cost":{"input":0.2,"output":0.5,"cache_read":0.1}},"mistral4-119b":{"id":"mistral4-119b","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.2,"output":0.6}},"kimi-k2.7-code-1100b":{"id":"kimi-k2.7-code-1100b","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.86,"output":3,"cache_read":0.18}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.15,"output":0.55}},"qwen3.6-35b":{"id":"qwen3.6-35b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.15,"output":0.5}}}},"trustedrouter":{"id":"trustedrouter","env":["TRUSTEDROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.trustedrouter.com/v1","name":"TrustedRouter","doc":"https://trustedrouter.com/docs","models":{"trustedrouter/zdr":{"id":"trustedrouter/zdr","name":"Zero Data Retention","description":"TrustedRouter privacy routing alias that prefers zero data retention model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/synth":{"id":"trustedrouter/synth","name":"Synth","description":"TrustedRouter synthesis orchestration alias that combines multiple model responses into one answer.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/e2e":{"id":"trustedrouter/e2e","name":"End-to-End Encrypted","description":"TrustedRouter privacy routing alias for end-to-end encrypted provider routes where available.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/synth-code":{"id":"trustedrouter/synth-code","name":"Synth Code","description":"TrustedRouter code synthesis orchestration alias that combines multiple model responses into one answer.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/fast":{"id":"trustedrouter/fast","name":"Fast","description":"TrustedRouter speed routing alias that prefers low-latency healthy model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/cheap":{"id":"trustedrouter/cheap","name":"Cheap","description":"TrustedRouter low-cost routing alias that prefers inexpensive healthy model endpoints.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}},"trustedrouter/auto":{"id":"trustedrouter/auto","name":"Auto","description":"TrustedRouter automatic routing alias that chooses a healthy supported model endpoint for the request.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-01","last_updated":"2026-06-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072}}}},"alibaba":{"id":"alibaba","env":["DASHSCOPE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://dashscope-intl.aliyuncs.com/compatible-mode/v1","name":"Alibaba","doc":"https://www.alibabacloud.com/help/en/model-studio/models","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen2-5-72b-instruct":{"id":"qwen2-5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":1.4,"output":5.6}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.2,"output":0.4,"cache_read":0.04}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":5}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":6}},"qwen2-5-omni-7b":{"id":"qwen2-5-omni-7b","name":"Qwen2.5-Omni 7B","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0.1,"output":0.4,"input_audio":6.76}},"qwen-mt-turbo":{"id":"qwen-mt-turbo","name":"Qwen-MT Turbo","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":0.16,"output":0.49}},"qwen-vl-max":{"id":"qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":3.2}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":2}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen3-14b":{"id":"qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.4,"reasoning":4.2}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen-vl-plus":{"id":"qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.21,"output":0.63}},"qwen-omni-turbo-realtime":{"id":"qwen-omni-turbo-realtime","name":"Qwen-Omni Turbo Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-05-08","last_updated":"2025-05-08","modalities":{"input":["text","image","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.27,"output":1.07,"input_audio":4.44,"output_audio":8.89}},"qwen3.5-27b":{"id":"qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.28,"cache_write":0}},"qwen-turbo":{"id":"qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384},"cost":{"input":0.05,"output":0.2,"reasoning":0.5}},"qwen3-livetranslate-flash-realtime":{"id":"qwen3-livetranslate-flash-realtime","name":"Qwen3-LiveTranslate Flash Realtime","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":10,"output":10,"input_audio":10,"output_audio":38}},"qwen3-vl-30b-a3b":{"id":"qwen3-vl-30b-a3b","name":"Qwen3-VL 30B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.8,"reasoning":2.4}},"qwen-vl-ocr":{"id":"qwen-vl-ocr","name":"Qwen-VL OCR","description":"OCR model for extracting structured text from documents and screenshots","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-10-28","last_updated":"2025-04-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":34096,"output":4096},"cost":{"input":0.72,"output":0.72}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwq-plus":{"id":"qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":2.4}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"reasoning":4.8}},"qwen2-5-7b-instruct":{"id":"qwen2-5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.175,"output":0.7}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.25,"tiers":[{"input":0.75,"output":3.75,"tier":{"type":"context","size":32000}},{"input":1.2,"output":6,"tier":{"type":"context","size":128000}}]}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3-asr-flash":{"id":"qwen3-asr-flash","name":"Qwen3-ASR Flash","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-04","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":53248,"output":4096},"cost":{"input":0.035,"output":0.035}},"qwen3-omni-flash":{"id":"qwen3-omni-flash","name":"Qwen3-Omni Flash","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.43,"output":1.66,"input_audio":3.81,"output_audio":15.11}},"qwen2-5-vl-72b-instruct":{"id":"qwen2-5-vl-72b-instruct","name":"Qwen2.5-VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.8,"output":8.4}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.2,"reasoning":4}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"qwen3-omni-flash-realtime":{"id":"qwen3-omni-flash-realtime","name":"Qwen3-Omni Flash Realtime","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":65536,"output":16384},"cost":{"input":0.52,"output":1.99,"input_audio":4.57,"output_audio":18.13}},"qwen2-5-vl-7b-instruct":{"id":"qwen2-5-vl-7b-instruct","name":"Qwen2.5-VL 7B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.05}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen2-5-14b-instruct":{"id":"qwen2-5-14b-instruct","name":"Qwen2.5 14B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.35,"output":1.4}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625}},"qwen-plus-character-ja":{"id":"qwen-plus-character-ja","name":"Qwen Plus Character (Japanese)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":512},"cost":{"input":0.5,"output":1.4}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen3-8b":{"id":"qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.18,"output":0.7,"reasoning":2.1}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-04","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.07,"output":0.27,"input_audio":4.44,"output_audio":8.89}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5,"tiers":[{"input":2.7,"output":13.5,"tier":{"type":"context","size":32000}},{"input":4.5,"output":22.5,"tier":{"type":"context","size":128000}}]}},"qwen2-5-32b-instruct":{"id":"qwen2-5-32b-instruct","name":"Qwen2.5 32B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-09","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.7,"output":2.8}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3-VL 235B-A22B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":2.8,"reasoning":8.4}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"reasoning":2.4}},"qwen-mt-plus":{"id":"qwen-mt-plus","name":"Qwen-MT Plus","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-01","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":8192},"cost":{"input":2.46,"output":7.37}},"qvq-max":{"id":"qvq-max","name":"QVQ Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qvq","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-25","last_updated":"2025-03-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":1.2,"output":4.8}}}},"nvidia":{"id":"nvidia","env":["NVIDIA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://integrate.api.nvidia.com/v1","name":"Nvidia","doc":"https://docs.api.nvidia.com/nim/","models":{"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0,"output":0}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen/qwen-image":{"id":"qwen/qwen-image","name":"Qwen Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":66536},"cost":{"input":0,"output":0}},"qwen/qwen-image-edit":{"id":"qwen/qwen-image-edit","name":"Qwen Image Edit","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-19","last_updated":"2025-08-19","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen/qwen2.5-coder-32b-instruct":{"id":"qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32b Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-06","last_updated":"2024-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"stepfun-ai/step-3.7-flash":{"id":"stepfun-ai/step-3.7-flash","name":"Step 3.7 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"stepfun-ai/step-3.5-flash":{"id":"stepfun-ai/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-pro-0813":{"id":"deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-flash-0731":{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"deepseek-ai/deepseek-v4-flash":{"id":"deepseek-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek-ai/deepseek-v4-pro":{"id":"deepseek-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-large-3-675b-instruct-2512":{"id":"mistralai/mistral-large-3-675b-instruct-2512","name":"Mistral Large 3 675B Instruct 2512","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"mistralai/mistral-nemotron":{"id":"mistralai/mistral-nemotron","name":"mistral-nemotron","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-11","last_updated":"2025-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":13108},"cost":{"input":0,"output":0}},"mistralai/ministral-14b-instruct-2512":{"id":"mistralai/ministral-14b-instruct-2512","name":"Ministral 3 14B Instruct 2512","description":"Compact Mistral VLM for chat and instruction-based workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-small-4-119b-2603":{"id":"mistralai/mistral-small-4-119b-2603","name":"mistral-small-4-119b-2603","description":"Efficient Mistral model for fast chat, extraction, and production assistants","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"mistralai/mixtral-8x7b-instruct":{"id":"mistralai/mixtral-8x7b-instruct","name":"Mistral: Mixtral 8x7B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2023-12-10","last_updated":"2026-03-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"mistralai/mistral-medium-3.5-128b":{"id":"mistralai/mistral-medium-3.5-128b","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"mistralai/mistral-7b-instruct-v0.3":{"id":"mistralai/mistral-7b-instruct-v0.3","name":"Mistral-7B-Instruct-v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0,"output":0}},"mistralai/mistral-medium-3-instruct":{"id":"mistralai/mistral-medium-3-instruct","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"input":131072,"output":32768},"cost":{"input":0,"output":0}},"mistralai/magistral-small-2506":{"id":"mistralai/magistral-small-2506","name":"Magistral Small 2506","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-09-25","last_updated":"2025-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"input":32768,"output":32768},"cost":{"input":0,"output":0}},"nvidia/streampetr":{"id":"nvidia/streampetr","name":"streampetr","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-3.3-nemotron-super-49b-v1.5":{"id":"nvidia/llama-3.3-nemotron-super-49b-v1.5","name":"Llama 3.3 Nemotron Super 49B v1.5","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"nvidia/usdcode":{"id":"nvidia/usdcode","name":"usdcode","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning","name":"Nemotron 3 Nano Omni","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":-1,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/cosmos-transfer1-7b":{"id":"nvidia/cosmos-transfer1-7b","name":"cosmos-transfer1-7b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-13","last_updated":"2025-06-30","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-voicechat":{"id":"nvidia/nemotron-voicechat","name":"nemotron-voicechat","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/studiovoice":{"id":"nvidia/studiovoice","name":"studiovoice","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-03","last_updated":"2025-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-content-safety":{"id":"nvidia/nemotron-3-content-safety","name":"nemotron-3-content-safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/cosmos-transfer2_5-2b":{"id":"nvidia/cosmos-transfer2_5-2b","name":"cosmos-transfer2.5-2b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/bevformer":{"id":"nvidia/bevformer","name":"bevformer","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-07-20","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-nano-vl-8b-v1":{"id":"nvidia/llama-3.1-nemotron-nano-vl-8b-v1","name":"Llama 3.1 Nemotron Nano VL 8B v1","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-10","last_updated":"2025-04-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"nvidia/magpie-tts-zeroshot":{"id":"nvidia/magpie-tts-zeroshot","name":"magpie-tts-zeroshot","description":"Speech generation model for controllable voice, narration, and audio delivery","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-22","last_updated":"2025-06-12","modalities":{"input":["text","audio"],"output":["audio"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-nano-12b-v2-vl":{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"Nemotron Nano 12B v2 VL","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0,"output":0}},"nvidia/sparsedrive":{"id":"nvidia/sparsedrive","name":"sparsedrive","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-07-20","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/llama-nemotron-embed-vl-1b-v2":{"id":"nvidia/llama-nemotron-embed-vl-1b-v2","name":"llama-nemotron-embed-vl-1b-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"nemotron","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-10","last_updated":"2026-02-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/synthetic-video-detector":{"id":"nvidia/synthetic-video-detector","name":"synthetic-video-detector","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.8}},"nvidia/llama-nemotron-rerank-vl-1b-v2":{"id":"nvidia/llama-nemotron-rerank-vl-1b-v2","name":"llama-nemotron-rerank-vl-1b-v2","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"nemotron","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/usdvalidate":{"id":"nvidia/usdvalidate","name":"usdvalidate","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-24","last_updated":"2025-01-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/active-speaker-detection":{"id":"nvidia/active-speaker-detection","name":"Active Speaker Detection","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["video"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-ultra-253b-v1":{"id":"nvidia/llama-3.1-nemotron-ultra-253b-v1","name":"Llama 3.1 Nemotron Ultra 253B","description":"Flagship Nemotron model for high-throughput reasoning and complex agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-04-07","last_updated":"2025-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"nvidia/llama-3_2-nemoretriever-300m-embed-v1":{"id":"nvidia/llama-3_2-nemoretriever-300m-embed-v1","name":"llama-3_2-nemoretriever-300m-embed-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-07-24","last_updated":"2025-07-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/nv-embedcode-7b-v1":{"id":"nvidia/nv-embedcode-7b-v1","name":"nv-embedcode-7b-v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-03-17","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-safety-guard-8b-v3":{"id":"nvidia/llama-3.1-nemotron-safety-guard-8b-v3","name":"llama-3.1-nemotron-safety-guard-8b-v3","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-mini-4b-instruct":{"id":"nvidia/nemotron-mini-4b-instruct","name":"nemotron-mini-4b-instruct","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-08-21","last_updated":"2024-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/cosmos-predict1-5b":{"id":"nvidia/cosmos-predict1-5b","name":"cosmos-predict1-5b","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-content-safety-reasoning-4b":{"id":"nvidia/nemotron-content-safety-reasoning-4b","name":"nemotron-content-safety-reasoning-4b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":false,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/riva-translate-4b-instruct-v1.1":{"id":"nvidia/riva-translate-4b-instruct-v1.1","name":"riva-translate-4b-instruct-v1_1","description":"Translation model for multilingual conversion, localization, and cross-language workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nvidia-nemotron-nano-9b-v2":{"id":"nvidia/nvidia-nemotron-nano-9b-v2","name":"nvidia-nemotron-nano-9b-v2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"nvidia/gliner-pii":{"id":"nvidia/gliner-pii","name":"gliner-pii","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-nano-8b-v1":{"id":"nvidia/llama-3.1-nemotron-nano-8b-v1","name":"Llama 3.1 Nemotron Nano 8B v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"nvidia/nv-embed-v1":{"id":"nvidia/nv-embed-v1","name":"nv-embed-v1","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-06-07","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":2048},"cost":{"input":0,"output":0}},"nvidia/rerank-qa-mistral-4b":{"id":"nvidia/rerank-qa-mistral-4b","name":"rerank-qa-mistral-4b","description":"Reranking model for improving retrieval quality in search and recommendation systems","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-03-17","last_updated":"2025-01-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.15}},"nvidia/nemotron-3.5-lightning-30b-a3b":{"id":"nvidia/nemotron-3.5-lightning-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"nvidia/llama-3.3-nemotron-super-49b-v1":{"id":"nvidia/llama-3.3-nemotron-super-49b-v1","name":"Llama 3.3 Nemotron Super 49B v1","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-04-07","last_updated":"2025-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"nemotron-3-nano-30b-a3b","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-12","last_updated":"2024-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"nvidia/llama-3.1-nemotron-70b-instruct":{"id":"nvidia/llama-3.1-nemotron-70b-instruct","name":"Llama 3.1 Nemotron 70B Instruct","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/cosmos-reason2-8b":{"id":"nvidia/cosmos-reason2-8b","name":"Cosmos Reason2 8B","description":"Vision language model for physical-world understanding with structured reasoning on video and images","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3n-e2b-it":{"id":"google/gemma-3n-e2b-it","name":"Gemma 3n E2b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-06-12","last_updated":"2025-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/google-paligemma":{"id":"google/google-paligemma","name":"paligemma","description":"Gemini multimodal model for text, image, audio, video, and document tasks","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-05-14","last_updated":"2024-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"google/gemma-2-2b-it":{"id":"google/gemma-2-2b-it","name":"Gemma 2 2b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-16","last_updated":"2024-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma-4-31B-IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0,"output":0}},"google/gemma-3n-e4b-it":{"id":"google/gemma-3n-e4b-it","name":"Gemma 3n E4b It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0,"output":0}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-guard-4-12b":{"id":"meta/llama-guard-4-12b","name":"Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"meta/llama-3.2-90b-vision-instruct":{"id":"meta/llama-3.2-90b-vision-instruct","name":"Llama-3.2-90B-Vision-Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-3.2-3b-instruct":{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0,"output":0}},"meta/llama-3.2-1b-instruct":{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/esmfold":{"id":"meta/esmfold","name":"esmfold","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-15","last_updated":"2025-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-4-maverick-17b-128e-instruct":{"id":"meta/llama-4-maverick-17b-128e-instruct","name":"Llama 4 Maverick 17b 128e Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-02","release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0,"output":0}},"meta/esm2-650m":{"id":"meta/esm2-650m","name":"esm2-650m","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-29","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"meta/llama-3.2-11b-vision-instruct":{"id":"meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11b Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.1-70b-instruct":{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-16","last_updated":"2024-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-3.3-70b-instruct":{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70b Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-26","last_updated":"2024-11-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"bytedance/seed-oss-36b-instruct":{"id":"bytedance/seed-oss-36b-instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0,"output":0}},"sarvamai/sarvam-m":{"id":"sarvamai/sarvam-m","name":"sarvam-m","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"microsoft/phi-4-multimodal-instruct":{"id":"microsoft/phi-4-multimodal-instruct","name":"Phi 4 Multimodal","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":16384},"cost":{"input":0,"output":0}},"microsoft/phi-4-mini-instruct":{"id":"microsoft/phi-4-mini-instruct","name":"Phi-4-Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0,"output":0}},"minimaxai/minimax-m2.7":{"id":"minimaxai/minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-04-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"minimaxai/minimax-m3":{"id":"minimaxai/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0,"output":0}},"baai/bge-m3":{"id":"baai/bge-m3","name":"BGE M3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"bge","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-01-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":1024},"cost":{"input":0,"output":0}},"abacusai/dracarys-llama-3.1-70b-instruct":{"id":"abacusai/dracarys-llama-3.1-70b-instruct","name":"dracarys-llama-3.1-70b-instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-11","last_updated":"2025-05-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2023-09","release_date":"2023-09-01","last_updated":"2025-09-05","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":0,"output":4096},"cost":{"input":0,"output":0}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS-120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-04","last_updated":"2025-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0}},"moonshotai/kimi-k2-instruct-0905":{"id":"moonshotai/kimi-k2-instruct-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}},"upstage/solar-10.7b-instruct":{"id":"upstage/solar-10.7b-instruct","name":"solar-10.7b-instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-06-05","last_updated":"2025-04-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"black-forest-labs/flux_1-kontext-dev":{"id":"black-forest-labs/flux_1-kontext-dev","name":"FLUX.1-Kontext-dev","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image"],"output":["image"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0,"output":0}},"black-forest-labs/flux_1-schnell":{"id":"black-forest-labs/flux_1-schnell","name":"FLUX.1-schnell","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2024-07","release_date":"2024-08-01","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":77,"input":77,"output":0},"cost":{"input":0,"output":0}},"black-forest-labs/flux_2-klein-4b":{"id":"black-forest-labs/flux_2-klein-4b","name":"FLUX.2 Klein 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-06","release_date":"2026-01-14","last_updated":"2026-01-31","modalities":{"input":["image","text"],"output":["image"]},"open_weights":true,"limit":{"context":40960,"output":40960},"cost":{"input":0,"output":0}},"black-forest-labs/flux.1-dev":{"id":"black-forest-labs/flux.1-dev","name":"FLUX.1-dev","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-08","release_date":"2024-08-01","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0},"cost":{"input":0,"output":0}}}},"jiekou":{"id":"jiekou","env":["JIEKOU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.jiekou.ai/openai","name":"Jiekou.AI","doc":"https://docs.jiekou.ai/docs/support/quickstart?utm_source=github_models.dev","models":{"gpt-5-nano":{"id":"gpt-5-nano","name":"gpt-5-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.045,"output":0.36}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"grok-4-1-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"gpt-5-codex":{"id":"gpt-5-codex","name":"gpt-5-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gpt-5-pro":{"id":"gpt-5-pro","name":"gpt-5-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":13.5,"output":108}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"gpt-5.1-codex-mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.225,"output":1.8}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"gpt-5.1-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"grok-code-fast-1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.18,"output":1.35}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"gpt-5.2-codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"gemini-2.5-pro-preview-06-05":{"id":"gemini-2.5-pro-preview-06-05","name":"gemini-2.5-pro-preview-06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":200000},"cost":{"input":1.125,"output":9}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"gemini-2.5-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09,"output":0.36}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"gpt-5.2-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":18.9,"output":151.2}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"gemini-3-pro-preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.8,"output":10.8}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"claude-opus-4-1-20250805","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":13.5,"output":67.5}},"gpt-5-chat-latest":{"id":"gpt-5-chat-latest","name":"gpt-5-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"grok-4-1-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"gemini-2.5-flash-lite-preview-06-17":{"id":"gemini-2.5-flash-lite-preview-06-17","name":"gemini-2.5-flash-lite-preview-06-17","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","video","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.09,"output":0.36}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"claude-opus-4-20250514","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":13.5,"output":67.5}},"gpt-5.1":{"id":"gpt-5.1","name":"gpt-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"gpt-5.1-codex-max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.125,"output":9}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"gemini-2.5-flash-lite-preview-09-2025","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.09,"output":0.36}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"claude-opus-4-6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"grok-4-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"claude-sonnet-4-5-20250929","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.7,"output":13.5}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"claude-haiku-4-5-20251001","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":20000,"output":64000},"cost":{"input":0.9,"output":4.5}},"grok-4-0709":{"id":"grok-4-0709","name":"grok-4-0709","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":2.7,"output":13.5}},"gemini-2.5-flash-preview-05-20":{"id":"gemini-2.5-flash-preview-05-20","name":"gemini-2.5-flash-preview-05-20","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":200000},"cost":{"input":0.135,"output":3.15}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"gemini-3-flash-preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.225,"output":1.8}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.125,"output":9}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.575,"output":12.6}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"claude-sonnet-4-20250514","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.7,"output":13.5}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.27,"output":2.25}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"grok-4-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.18,"output":0.45}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":1.1,"output":4.4}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"claude-opus-4-5-20251101","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65536},"cost":{"input":4.5,"output":22.5}},"o3":{"id":"o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":10,"output":40}},"qwen/qwen3-30b-a3b-fp8":{"id":"qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.09,"output":0.45}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22b Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":3}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.15,"output":1.5}},"qwen/qwen3-235b-a22b-fp8":{"id":"qwen/qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.2,"output":0.8}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"qwen/qwen3-coder-next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"qwen/qwen3-32b-fp8":{"id":"qwen/qwen3-32b-fp8","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":20000},"cost":{"input":0.1,"output":0.45}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.15,"output":0.8}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.2}},"baidu/ernie-4.5-300b-a47b-paddle":{"id":"baidu/ernie-4.5-300b-a47b-paddle","name":"ERNIE 4.5 300B A47B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":12000},"cost":{"input":0.28,"output":1.1}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"Minimax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":131071}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"zai-org/glm-4.7":{"id":"zai-org/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"zai-org/glm-4.5":{"id":"zai-org/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2}},"zai-org/glm-4.5v":{"id":"zai-org/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glmv","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8}},"zai-org/glm-4.7-flash":{"id":"zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4}},"minimaxai/minimax-m1-80k":{"id":"minimaxai/minimax-m1-80k","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-v3-0324":{"id":"deepseek/deepseek-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.28,"output":1.14}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32767}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1}},"xiaomimimo/mimo-v2-flash":{"id":"xiaomimimo/mimo-v2-flash","name":"XiaomiMiMo/MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2-instruct":{"id":"moonshotai/kimi-k2-instruct","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":262143}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3}}}},"frogbot":{"id":"frogbot","env":["FROGBOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://app.frogbot.ai/api/v1","name":"FrogBot","doc":"https://docs.frogbot.ai","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"gpt-5-4-nano":{"id":"gpt-5-4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"qwen-3-6-plus":{"id":"qwen-3-6-plus","name":"Qwen 3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"gpt-5-3-codex":{"id":"gpt-5-3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.2}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2-5":{"id":"minimax-m2-5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-01-15","last_updated":"2025-02-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-1-pro-preview":{"id":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"zai-glm-5-1":{"id":"zai-glm-5-1","name":"Z.AI GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-01-20","last_updated":"2025-02-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":8192},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek v4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":1.74,"output":3.48,"cache_read":0.14}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.31}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":192000,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-07-17","last_updated":"2025-07-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}}}},"ovhcloud":{"id":"ovhcloud","env":["OVHCLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://oai.endpoints.kepler.ai.cloud.ovh.net/v1","name":"OVHcloud AI Endpoints","doc":"https://www.ovhcloud.com/en/public-cloud/ai-endpoints/catalog//","models":{"qwen3guard-gen-0.6b":{"id":"qwen3guard-gen-0.6b","name":"Qwen3Guard-Gen-0.6B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5-9B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.12,"output":0.18}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8-27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3-32B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-16","last_updated":"2025-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.09,"output":0.25}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.18}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"Qwen2.5-VL-72B-Instruct","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-31","last_updated":"2025-03-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":1.01,"output":1.01}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder-30B-A3B-Instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.07,"output":0.26}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5-397B-A17B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-18","last_updated":"2026-05-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.71,"output":4.25}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6-27B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.47,"output":3.19}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"Mistral-Nemo-Instruct-2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.14,"output":0.14}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"Mistral-Small-3.2-24B-Instruct-2506","description":"Efficient Mistral model for fast chat, extraction, and production assistants","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-16","last_updated":"2025-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.31}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"Mistral-7B-Instruct-v0.3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.11,"output":0.11}},"qwen3guard-gen-8b":{"id":"qwen3guard-gen-8b","name":"Qwen3Guard-Gen-8B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.09,"output":0.47}},"meta-llama-3_3-70b-instruct":{"id":"meta-llama-3_3-70b-instruct","name":"Meta-Llama-3_3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-01","last_updated":"2025-04-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.74,"output":0.74}}}},"xpersona":{"id":"xpersona","env":["XPERSONA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://www.xpersona.co/v1","name":"Xpersona","doc":"https://www.xpersona.co/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.9,"output":5.55,"reasoning":5.55,"cache_read":0.09}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}},"xpersona-gpt-5.5":{"id":"xpersona-gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-30","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":18,"reasoning":18,"cache_read":0.3}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":0.75,"output":6,"reasoning":6,"cache_read":0.075}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.55,"output":12.2,"reasoning":12.2,"cache_read":0.155}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":18.5,"reasoning":18.5,"cache_read":0.3}},"xpersona-frieren-coder":{"id":"xpersona-frieren-coder","name":"Xpersona Frieren 1","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-30","release_date":"2026-05-01","last_updated":"2026-05-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":1.5,"output":6,"reasoning":6,"cache_read":0.15}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":0.375,"output":4,"reasoning":4,"cache_read":0.0375}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.6,"output":3.7,"reasoning":3.7,"cache_read":0.06}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.5,"output":9.25,"reasoning":9.25,"cache_read":0.15}},"gpt-5.6":{"id":"gpt-5.6","name":"GPT-5.6","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":372000,"output":128000},"cost":{"input":1.5,"output":2,"reasoning":2,"cache_read":0.15}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":1.5,"output":12,"reasoning":12,"cache_read":0.15}}}},"anthropic":{"id":"anthropic","env":["ANTHROPIC_API_KEY"],"npm":"@ai-sdk/anthropic","name":"Anthropic","doc":"https://docs.anthropic.com/en/docs/about-claude/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-04","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-14","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-07","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-29","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}}}},"google":{"id":"google","env":["GOOGLE_API_KEY","GOOGLE_GENERATIVE_AI_API_KEY","GEMINI_API_KEY"],"npm":"@ai-sdk/google","name":"Google","doc":"https://ai.google.dev/gemini-api/docs/models","models":{"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"gemini-3.1-pro-preview-customtools":{"id":"gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3.1-flash-lite-image":{"id":"gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.25,"output":30}},"lyria-3-clip-preview":{"id":"lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Music generation model for short 30-second clips, loops, and previews from text or image prompts","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30,"cache_read":0.075}},"deep-research-max-preview-04-2026":{"id":"deep-research-max-preview-04-2026","name":"Deep Research Max Preview (Apr-21-2026)","description":"Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":120}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"deep-research-preview-04-2026":{"id":"deep-research-preview-04-2026","name":"Deep Research Preview (Apr-21-2026)","description":"Agentic model for autonomous multi-step research, synthesis, and cited reports","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"gemini-2.5-computer-use-preview-10-2025":{"id":"gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview 10-2025","description":"Specialized Gemini 2.5 model for browser-control agents that automate UI tasks","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1.25,"output":10,"tiers":[{"input":2.5,"output":15,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"gemini-3.1-flash-live-preview":{"id":"gemini-3.1-flash-live-preview","name":"Gemini 3.1 Flash Live Preview","description":"High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-26","last_updated":"2026-03-26","modalities":{"input":["text","image","video","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.75,"output":4.5,"input_audio":3,"output_audio":12}},"gemini-2.5-pro-preview-tts":{"id":"gemini-2.5-pro-preview-tts","name":"Gemini 2.5 Pro Preview TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":1,"output":20}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"veo-3.1-generate-preview":{"id":"veo-3.1-generate-preview","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-15","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192},"status":"beta"},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Legacy model retained for compatibility with older integrations","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"veo-3.1-fast-generate-preview":{"id":"veo-3.1-fast-generate-preview","name":"Veo 3.1 fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-10-15","last_updated":"2026-01-01","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192}},"gemini-2.5-flash-preview-tts":{"id":"gemini-2.5-flash-preview-tts","name":"Gemini 2.5 Flash Preview TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":0.5,"output":10}},"gemini-embedding-001":{"id":"gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":1},"cost":{"input":0.15,"output":0}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","video","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":60}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":120}},"gemini-3.1-flash-tts-preview":{"id":"gemini-3.1-flash-tts-preview","name":"Gemini 3.1 Flash TTS Preview","description":"Low-latency speech generation with steerable prompts and expressive audio tags","family":"gemini-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":8192,"output":16384},"cost":{"input":1,"output":20}},"veo-3.1-lite-generate-preview":{"id":"veo-3.1-lite-generate-preview","name":"Veo 3.1 lite","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":480,"output":8192}},"gemini-flash-lite-latest":{"id":"gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768}},"gemini-embedding-2":{"id":"gemini-embedding-2","name":"Gemini Embedding 2","description":"Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space","family":"gemini","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1},"cost":{"input":0.2,"output":0,"input_audio":6.5}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.5-live-translate-preview":{"id":"gemini-3.5-live-translate-preview","name":"Gemini 3.5 Live Translate Preview","description":"Low-latency audio-to-audio model for real-time speech translation across 70+ languages","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["audio"],"output":["audio","text"]},"open_weights":false,"limit":{"context":16384,"output":32768},"cost":{"input":3.5,"output":21,"input_audio":3.5,"output_audio":21}},"lyria-3-pro-preview":{"id":"lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Music generation model for full-length songs from text or images with vocals and structure","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gemini-flash-latest":{"id":"gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"input_audio":0.75}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":60}},"gemini-omni-flash-preview":{"id":"gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","description":"Video generation and editing model for fast, conversational text- and image-to-video workflows","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","video"],"output":["video"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":1.5,"output":17.5}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}}}},"baseten":{"id":"baseten","env":["BASETEN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.baseten.co/v1","name":"Baseten","doc":"https://docs.baseten.co/inference/model-apis/overview","models":{"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":131000},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.32,"output":3.96}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","name":"Nemotron Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/Nemotron-120B-A12B":{"id":"nvidia/Nemotron-120B-A12B","name":"Nemotron Super","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.3,"output":0.75,"cache_read":0.06}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":1.3,"output":4.3,"cache_read":0.26}},"zai-org/GLM-5.2-Fast":{"id":"zai-org/GLM-5.2-Fast","name":"GLM 5.2 Fast","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.3}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":0.6,"output":2.2,"cache_read":0.12}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":202800},"cost":{"input":0.95,"output":3.15,"cache_read":0.2}},"zai-org/GLM-5.3-Fast":{"id":"zai-org/GLM-5.3-Fast","name":"GLM 5.3 Fast","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1,"output":4.05}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204000,"output":204000},"status":"deprecated","cost":{"input":0.3,"output":1.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"OpenAI GPT 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128072,"output":128072},"cost":{"input":0.1,"output":0.5}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-01-30","last_updated":"2026-02-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.6,"output":3,"cache_read":0.12}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":3,"output":15}}}},"vercel":{"id":"vercel","env":["AI_GATEWAY_API_KEY"],"npm":"@ai-sdk/gateway","name":"Vercel AI Gateway","doc":"https://github.com/vercel/ai/tree/5eb85cc45a259553501f535b8ac79a77d0e79223/packages/gateway","models":{"voyage/voyage-code-3":{"id":"voyage/voyage-code-3","name":"voyage-code-3","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-12-04","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-3.5":{"id":"voyage/voyage-3.5","name":"voyage-3.5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-3.5-lite":{"id":"voyage/voyage-3.5-lite","name":"voyage-3.5-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-3-large":{"id":"voyage/voyage-3-large","name":"voyage-3-large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-01-07","last_updated":"2024-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-code-2":{"id":"voyage/voyage-code-2","name":"voyage-code-2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-4":{"id":"voyage/voyage-4","name":"voyage-4","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"voyage/voyage-finance-2":{"id":"voyage/voyage-finance-2","name":"voyage-finance-2","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-06-03","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/rerank-2.5":{"id":"voyage/rerank-2.5","name":"Voyage Rerank 2.5","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"voyage/voyage-law-2":{"id":"voyage/voyage-law-2","name":"voyage-law-2","description":"General-purpose chat model for instruction following, writing, and analysis","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-15","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"voyage/voyage-4-large":{"id":"voyage/voyage-4-large","name":"voyage-4-large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"voyage/rerank-2.5-lite":{"id":"voyage/rerank-2.5-lite","name":"Voyage Rerank 2.5 Lite","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"voyage/voyage-4-lite":{"id":"voyage/voyage-4-lite","name":"voyage-4-lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"voyage","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph v3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":0.8,"output":1.2}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph v3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-08-15","last_updated":"2024-08-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.9,"output":1.9}},"poolside/laguna-s-2.1-free":{"id":"poolside/laguna-s-2.1-free","name":"Laguna S 2.1 Free","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"klingai/kling-v2.5-turbo-i2v":{"id":"klingai/kling-v2.5-turbo-i2v","name":"Kling v2.5 Turbo Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-motion-control":{"id":"klingai/kling-v3.0-motion-control","name":"Kling v3.0 Motion Control","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-t2v":{"id":"klingai/kling-v2.6-t2v","name":"Kling v2.6 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.5-turbo-t2v":{"id":"klingai/kling-v2.5-turbo-t2v","name":"Kling v2.5 Turbo Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-t2v":{"id":"klingai/kling-v3.0-t2v","name":"Kling v3.0 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-i2v":{"id":"klingai/kling-v2.6-i2v","name":"Kling v2.6 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v2.6-motion-control":{"id":"klingai/kling-v2.6-motion-control","name":"Kling v2.6 Motion Control","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-21","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"klingai/kling-v3.0-i2v":{"id":"klingai/kling-v3.0-i2v","name":"Kling v3.0 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"ling","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"kwaipilot/kat-coder-pro-v1":{"id":"kwaipilot/kat-coder-pro-v1","name":"KAT-Coder-Pro V1","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-11-09","last_updated":"2025-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kwaipilot/kat-coder-pro-v2":{"id":"kwaipilot/kat-coder-pro-v2","name":"Kat Coder Pro V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"Kat Coder Pro V2.5","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":80000},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"kwaipilot/kat-coder-air-v2.5":{"id":"kwaipilot/kat-coder-air-v2.5","name":"Kat Coder Air V2.5","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":80000},"cost":{"input":0.15,"output":0.6,"cache_read":0.03}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"StepFun 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262114,"output":262114},"cost":{"input":0.09,"output":0.3,"cache_read":0.02}},"interfaze/interfaze-beta":{"id":"interfaze/interfaze-beta","name":"Interfaze Beta","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"release_date":"2025-10-07","last_updated":"2026-04-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.5,"output":3.5}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo M2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131100},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131000},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-23","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-h3":{"id":"minimax/minimax-h3","name":"MiniMax H3","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 High Speed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131100},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"Minimax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 High Speed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131000},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-h3-max":{"id":"minimax/minimax-h3-max","name":"MiniMax H3 Max","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen 3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"alibaba/qwen3-vl-thinking":{"id":"alibaba/qwen3-vl-thinking","name":"Qwen3 VL Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-09-23","last_updated":"2025-09-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"alibaba/qwen-3-235b":{"id":"alibaba/qwen-3-235b","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.88}},"alibaba/qwen3-coder-plus":{"id":"alibaba/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2}},"alibaba/qwen3-next-80b-a3b-thinking":{"id":"alibaba/qwen3-next-80b-a3b-thinking","name":"Qwen3 Next 80B A3B Thinking","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"alibaba/qwen3.8-flash-next":{"id":"alibaba/qwen3.8-flash-next","name":"Qwen 3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.12,"output":0.4,"cache_read":0.01}},"alibaba/qwen3-coder-30b-a3b":{"id":"alibaba/qwen3-coder-30b-a3b","name":"Qwen 3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-31","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"alibaba/qwen3-next-80b-a3b-instruct":{"id":"alibaba/qwen3-next-80b-a3b-instruct","name":"Qwen3 Next 80B A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"alibaba/wan-v3.0-video":{"id":"alibaba/wan-v3.0-video","name":"Wan v3.0 Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-23","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.6-plus":{"id":"alibaba/qwen3.6-plus","name":"Qwen 3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"alibaba/wan-v2.7-r2v":{"id":"alibaba/wan-v2.7-r2v","name":"Wan v2.7 Reference-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-27b":{"id":"alibaba/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"alibaba/wan-v2.6-t2v":{"id":"alibaba/wan-v2.6-t2v","name":"Wan v2.6 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v3.0-video-prime":{"id":"alibaba/wan-v3.0-video-prime","name":"Wan v3.0 Video Prime","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen-3-14b":{"id":"alibaba/qwen-3-14b","name":"Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.24}},"alibaba/qwen3-vl-instruct":{"id":"alibaba/qwen3-vl-instruct","name":"Qwen3 VL Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":129024},"cost":{"input":0.4,"output":1.6}},"alibaba/qwen3-235b-a22b-thinking":{"id":"alibaba/qwen3-235b-a22b-thinking","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"alibaba/qwen3.5-flash":{"id":"alibaba/qwen3.5-flash","name":"Qwen 3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.001,"cache_write":0.125}},"alibaba/qwen3-coder":{"id":"alibaba/qwen3-coder","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-22","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.3}},"alibaba/qwen3-coder-next":{"id":"alibaba/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.2}},"alibaba/qwen3-embedding-4b":{"id":"alibaba/qwen3-embedding-4b","name":"Qwen3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3-max-preview":{"id":"alibaba/qwen3-max-preview","name":"Qwen3 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-05","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.2,"output":6,"cache_read":0.24}},"alibaba/qwen3-embedding-8b":{"id":"alibaba/qwen3-embedding-8b","name":"Qwen3 Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3.6-27b":{"id":"alibaba/qwen3.6-27b","name":"Qwen 3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3.6}},"alibaba/wan-v2.6-r2v":{"id":"alibaba/wan-v2.6-r2v","name":"Wan v2.6 Reference-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.7-flash":{"id":"alibaba/qwen3.7-flash","name":"Qwen 3.7 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":64000},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"alibaba/qwen3-max-thinking":{"id":"alibaba/qwen3-max-thinking","name":"Qwen 3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-23","last_updated":"2025-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24}},"alibaba/qwen3.8-max-0902":{"id":"alibaba/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":1.2,"output":6,"cache_read":0.24}},"alibaba/wan-v2.6-i2v-flash":{"id":"alibaba/wan-v2.6-i2v-flash","name":"Wan v2.6 Image-to-Video Flash","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-2.4t-a95b":{"id":"alibaba/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/wan-v2.6-r2v-flash":{"id":"alibaba/wan-v2.6-r2v-flash","name":"Wan v2.6 Reference-to-Video Flash","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/wan-v2.5-t2v-preview":{"id":"alibaba/wan-v2.5-t2v-preview","name":"Wan v2.5 Text-to-Video Preview","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen-3-30b":{"id":"alibaba/qwen-3-30b","name":"Qwen3-30B-A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"alibaba/qwen3.8-flash":{"id":"alibaba/qwen3.8-flash","name":"Qwen 3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":991000,"output":128000},"cost":{"input":0.16,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"alibaba/wan-v2.6-i2v":{"id":"alibaba/wan-v2.6-i2v","name":"Wan v2.6 Image-to-Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen 3.8 Max","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"alibaba/qwen-3.6-max-preview":{"id":"alibaba/qwen-3.6-max-preview","name":"Qwen 3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":240000,"output":64000},"cost":{"input":1.3,"output":7.8,"cache_read":0.26,"cache_write":1.625}},"alibaba/qwen3-embedding-0.6b":{"id":"alibaba/qwen3-embedding-0.6b","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768}},"alibaba/qwen3-vl-235b-a22b-instruct":{"id":"alibaba/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":129024},"cost":{"input":0.4,"output":1.6}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen 3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"alibaba/qwen-3-32b":{"id":"alibaba/qwen-3-32b","name":"Qwen 3.32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":38912}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-28","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.16,"output":0.64}},"alibaba/wan-v2.7-t2v":{"id":"alibaba/wan-v2.7-t2v","name":"Wan v2.7 Text-to-Video","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"alibaba/qwen3.5-plus":{"id":"alibaba/qwen3.5-plus","name":"Qwen 3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}},"nvidia/nemotron-nano-9b-v2":{"id":"nvidia/nemotron-nano-9b-v2","name":"Nvidia Nemotron Nano 9B V2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.06,"output":0.23}},"nvidia/nemotron-nano-12b-v2-vl":{"id":"nvidia/nemotron-nano-12b-v2-vl","name":"Nvidia Nemotron Nano 12B V2 VL","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-10-28","last_updated":"2025-10-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.2,"output":0.6}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"NVIDIA Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":0.15,"output":0.65}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.24}},"spacexai/grok-voice-think-fast-2.0":{"id":"spacexai/grok-voice-think-fast-2.0","name":"Grok Voice Think Fast 2.0","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-29","last_updated":"2026-07-29","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-reasoning":{"id":"spacexai/grok-4.20-reasoning","name":"Grok 4.20 Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"spacexai/grok-4.20-multi-agent":{"id":"spacexai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"spacexai/grok-4.3":{"id":"spacexai/grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"spacexai/grok-4.20-reasoning-beta":{"id":"spacexai/grok-4.20-reasoning-beta","name":"Grok 4.20 Beta Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"spacexai/grok-imagine-image":{"id":"spacexai/grok-imagine-image","name":"Grok Imagine Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-tts":{"id":"spacexai/grok-tts","name":"Grok TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.20-non-reasoning":{"id":"spacexai/grok-4.20-non-reasoning","name":"Grok 4.20 Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2026-03-10","last_updated":"2026-03-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"spacexai/grok-imagine-video":{"id":"spacexai/grok-imagine-video","name":"Grok Imagine","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.5":{"id":"spacexai/grok-4.5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"spacexai/grok-build-0.1":{"id":"spacexai/grok-build-0.1","name":"Grok Build 0.1","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"spacexai/grok-4.20-multi-agent-beta":{"id":"spacexai/grok-4.20-multi-agent-beta","name":"Grok 4.20 Multi Agent Beta","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"spacexai/grok-4.20-non-reasoning-beta":{"id":"spacexai/grok-4.20-non-reasoning-beta","name":"Grok 4.20 Beta Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.4}},"spacexai/grok-4.1-fast-non-reasoning":{"id":"spacexai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"spacexai/grok-4.1-fast-reasoning":{"id":"spacexai/grok-4.1-fast-reasoning","name":"Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"spacexai/grok-imagine-video-1.5":{"id":"spacexai/grok-imagine-video-1.5","name":"Grok Imagine Video 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-30","last_updated":"2026-05-30","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-imagine-image-2.0":{"id":"spacexai/grok-imagine-image-2.0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-4.6":{"id":"spacexai/grok-4.6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"spacexai/grok-stt":{"id":"spacexai/grok-stt","name":"Grok STT","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"spacexai/grok-voice-think-fast-1.0":{"id":"spacexai/grok-voice-think-fast-1.0","name":"Grok Voice Think Fast 1.0","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"grok","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"prodia/flux-fast-schnell":{"id":"prodia/flux-fast-schnell","name":"Flux Schnell","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-02","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5-fast":{"id":"anthropic/claude-opus-5-fast","name":"Claude Opus 5 (Fast)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Claude Haiku 3","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.8-fast":{"id":"anthropic/claude-opus-4.8-fast","name":"Claude Opus 4.8 (Fast)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":8192},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Gemini 3.1 Flash Lite Image (Nano Banana 2 Lite)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5,"cache_read":0.03}},"google/gemini-3.5-transcribe-live":{"id":"google/gemini-3.5-transcribe-live","name":"Gemini 3.5 Transcribe Live","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana (Gemini 2.5 Flash Image)","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-3.5-transcribe":{"id":"google/gemini-3.5-transcribe","name":"Gemini 3.5 Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2,"output":12}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/veo-3.1-fast-generate-001":{"id":"google/veo-3.1-fast-generate-001","name":"Veo 3.1 Fast Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.25,"output":1.5,"cache_read":0.03}},"google/text-embedding-005":{"id":"google/text-embedding-005","name":"Text Embedding 005","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-08-01","last_updated":"2024-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/veo-3.0-generate-001":{"id":"google/veo-3.0-generate-001","name":"Veo 3.0","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-20","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/veo-3.1-generate-001":{"id":"google/veo-3.1-generate-001","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"google/gemini-embedding-2":{"id":"google/gemini-embedding-2","name":"Gemini Embedding 2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-11","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/veo-3.0-fast-generate-001":{"id":"google/veo-3.0-fast-generate-001","name":"Veo 3.0 Fast Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-07-31","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/text-multilingual-embedding-002":{"id":"google/text-multilingual-embedding-002","name":"Text Multilingual Embedding 002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-01","last_updated":"2024-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Gemini 3.1 Flash Image Preview (Nano Banana 2)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-3-flash":{"id":"google/gemini-3-flash","name":"Gemini 3 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"google/gemini-omni-flash-preview":{"id":"google/gemini-omni-flash-preview","name":"Gemini Omni Flash Preview","description":"Omni-modal model for text, vision, audio, and multimodal agent tasks","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":57920},"cost":{"input":1.5,"output":9}},"google/veo-3.1-lite-generate-001":{"id":"google/veo-3.1-lite-generate-001","name":"Veo 3.1 Lite Generate","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"bfl/flux-kontext-max":{"id":"bfl/flux-kontext-max","name":"FLUX.1 Kontext Max","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-2-pro":{"id":"bfl/flux-2-pro","name":"FLUX.2 [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":67300,"output":67300}},"bfl/flux-2-max":{"id":"bfl/flux-2-max","name":"FLUX.2 [max]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":67300,"output":67300}},"bfl/flux-pro-1.1-ultra":{"id":"bfl/flux-pro-1.1-ultra","name":"FLUX1.1 [pro] Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-11-01","last_updated":"2024-11","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-kontext-pro":{"id":"bfl/flux-kontext-pro","name":"FLUX.1 Kontext Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-29","last_updated":"2025-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-3-video":{"id":"bfl/flux-3-video","name":"Flux 3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-04","last_updated":"2026-08-04","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-pro-1.0-fill":{"id":"bfl/flux-pro-1.0-fill","name":"FLUX.1 Fill [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-01","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"bfl/flux-2-flex":{"id":"bfl/flux-2-flex","name":"FLUX.2 [flex]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-klein-9b":{"id":"bfl/flux-2-klein-9b","name":"FLUX.2 [klein] 9B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-2-klein-4b":{"id":"bfl/flux-2-klein-4b","name":"FLUX.2 [klein] 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-15","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bfl/flux-pro-1.1":{"id":"bfl/flux-pro-1.1","name":"FLUX1.1 [pro]","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-02","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/llama-3.1-8b":{"id":"meta/llama-3.1-8b","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.22,"output":0.22}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"muse","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"meta/llama-3.1-70b":{"id":"meta/llama-3.1-70b","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.72,"output":0.72}},"meta/muse-image-1.0":{"id":"meta/muse-image-1.0","name":"Muse Image 1.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"muse","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"meta/llama-3.3-70b":{"id":"meta/llama-3.3-70b","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-4-scout":{"id":"meta/llama-4-scout","name":"Llama-4-Scout-17B-16E-Instruct-FP8","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"meta/llama-4-maverick":{"id":"meta/llama-4-maverick","name":"Llama-4-Maverick-17B-128E-Instruct-FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0,"output":0}},"quiverai/arrow-1.1":{"id":"quiverai/arrow-1.1","name":"Arrow 1.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":131072,"output":131072}},"bytedance/seedance-2.0-mini":{"id":"bytedance/seedance-2.0-mini","name":"Seedance 2.0 Mini","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-06-22","last_updated":"2026-06-22","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.0-pro-fast":{"id":"bytedance/seedance-v1.0-pro-fast","name":"Seedance v1.0 Pro Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-24","last_updated":"2025-10-31","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.0-pro":{"id":"bytedance/seedance-v1.0-pro","name":"Seedance v1.0 Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-06-11","last_updated":"2025-06-11","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-v1.5-pro":{"id":"bytedance/seedance-v1.5-pro","name":"Seedance v1.5 Pro","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-4.5":{"id":"bytedance/seedream-4.5","name":"Seedream 4.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-11-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seed-1.8":{"id":"bytedance/seed-1.8","name":"Seed 1.8","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-01","last_updated":"2025-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/seed-1.6":{"id":"bytedance/seed-1.6","name":"Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-01","last_updated":"2025-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"bytedance/seedance-2.0-fast":{"id":"bytedance/seedance-2.0-fast","name":"Seedance 2.0 Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-4.0":{"id":"bytedance/seedream-4.0","name":"Seedream 4.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-09-09","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-5.0-lite":{"id":"bytedance/seedream-5.0-lite","name":"Seedream 5.0 Lite","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.5":{"id":"bytedance/seedance-2.5","name":"Seedance 2.5","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedance-2.0":{"id":"bytedance/seedance-2.0","name":"Seedance 2.0","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-14","last_updated":"2026-04-14","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0}},"bytedance/seedream-5.0-pro":{"id":"bytedance/seedream-5.0-pro","name":"Seedream 5.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-11","last_updated":"2026-07-11","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"inception/mercury-coder-small":{"id":"inception/mercury-coder-small","name":"Mercury Coder Small Beta","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"mercury","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-02-26","last_updated":"2025-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":16384},"cost":{"input":0.25,"output":1}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-24","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.25,"output":0.75,"cache_read":0.024999999999999998}},"fish-audio/transcribe-1":{"id":"fish-audio/transcribe-1","name":"Transcribe-1","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2.1-pro-free":{"id":"fish-audio/s2.1-pro-free","name":"S2.1 Pro (Free)","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-28","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2.1-pro":{"id":"fish-audio/s2.1-pro","name":"S2.1 Pro","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-07-28","last_updated":"2026-07-28","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s1-free":{"id":"fish-audio/s1-free","name":"S1 (Free)","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2-pro":{"id":"fish-audio/s2-pro","name":"S2 Pro","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s2-pro-free":{"id":"fish-audio/s2-pro-free","name":"S2 Pro (Free)","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/transcribe-1-free":{"id":"fish-audio/transcribe-1-free","name":"Transcribe-1 (Free)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"fish-audio/s1":{"id":"fish-audio/s1","name":"S1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/namazu":{"id":"sakana/namazu","name":"Sakana Namazu","description":"Multi-agent model for routing expert agents across complex analytical tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.066}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.076,"output":0.153,"cache_read":0.014}},"deepseek/deepseek-v3.2-thinking":{"id":"deepseek/deepseek-v3.2-thinking","name":"DeepSeek V3.2 Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8000},"cost":{"input":0.62,"output":1.85}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8000},"cost":{"input":0.62,"output":1.85}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":128000},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"amazon/nova-2-lite":{"id":"amazon/nova-2-lite","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2024-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}},"amazon/titan-embed-text-v2":{"id":"amazon/titan-embed-text-v2","name":"Titan Text Embeddings V2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"titan-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-30","last_updated":"2024-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"amazon/nova-pro":{"id":"amazon/nova-pro","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"amazon/nova-lite":{"id":"amazon/nova-lite","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"amazon/nova-micro":{"id":"amazon/nova-micro","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"Ling 3.0 Flash Fin","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante":{"id":"inclusionai/ling-3.0-flash-sante","name":"Ling 3.0 Flash Sante","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante-free":{"id":"inclusionai/ling-3.0-flash-sante-free","name":"Ling 3.0 Flash Sante (Free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-fin-free":{"id":"inclusionai/ling-3.0-flash-fin-free","name":"Ling 3.0 Flash Fin (Free)","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0}},"openai/gpt-5-fast":{"id":"openai/gpt-5-fast","name":"GPT-5 (Fast)","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":128000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-transcribe":{"id":"openai/gpt-4o-mini-transcribe","name":"GPT-4o mini Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"o-mini","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":1.25,"output":5}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.5-fast":{"id":"openai/gpt-5.5-fast","name":"GPT 5.5 (Fast)","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":12.5,"output":75,"cache_read":1.25}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.4-mini-fast":{"id":"openai/gpt-5.4-mini-fast","name":"GPT 5.4 Mini (Fast)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT 5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-realtime-1.5":{"id":"openai/gpt-realtime-1.5","name":"GPT-Realtime-1.5","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":4,"output":16,"cache_read":0.4}},"openai/gpt-5-mini-fast":{"id":"openai/gpt-5-mini-fast","name":"GPT-5 mini (Fast)","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.45,"output":3.6,"cache_read":0.045}},"openai/tts-1":{"id":"openai/tts-1","name":"TTS-1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-6-astra-fast":{"id":"openai/gpt-6-astra-fast","name":"GPT-6 Astra (Fast)","description":"Fast variant of GPT-6 Astra for low-latency assistance and high-volume workloads.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":20,"output":100,"cache_read":2,"cache_write":25,"tiers":[{"input":40,"output":150,"cache_read":4,"cache_write":25,"tier":{"type":"context","size":272001}}],"context_over_200k":{"input":40,"output":150,"cache_read":4,"cache_write":25}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT 5.2 ","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-realtime-2":{"id":"openai/gpt-realtime-2","name":"gpt-realtime-2","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT 5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"input":122880,"output":8192},"cost":{"input":0.05,"output":0.2}},"openai/gpt-4o-transcribe":{"id":"openai/gpt-4o-transcribe","name":"GPT-4o Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2.5,"output":10}},"openai/o4-mini-fast":{"id":"openai/o4-mini-fast","name":"o4-mini (Fast)","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"gpt-oss-safeguard-20b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-10-29","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":112000,"output":16000},"cost":{"input":0.07,"output":0.2}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT 5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-oss-safeguard-120b":{"id":"openai/gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"input":112000,"output":16000},"cost":{"input":0.15,"output":0.6}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT 5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-4.1-fast":{"id":"openai/gpt-4.1-fast","name":"GPT-4.1 (Fast)","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":3.5,"output":14,"cache_read":0.875}},"openai/gpt-4o-mini-fast":{"id":"openai/gpt-4o-mini-fast","name":"GPT-4o mini (Fast)","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":0.25,"output":1,"cache_read":0.125}},"openai/gpt-5.6-luna-fast":{"id":"openai/gpt-5.6-luna-fast","name":"GPT 5.6 Luna (Fast)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT 5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-image-1.5":{"id":"openai/gpt-image-1.5","name":"GPT Image 1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":32,"cache_read":1.25}},"openai/gpt-5.4-fast":{"id":"openai/gpt-5.4-fast","name":"GPT 5.4 (Fast)","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.1-thinking-fast":{"id":"openai/gpt-5.1-thinking-fast","name":"GPT 5.1 Thinking (Fast)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2025-11-12","last_updated":"2025-11-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.5,"output":20,"cache_read":0.25}},"openai/text-embedding-ada-002":{"id":"openai/text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-image-1":{"id":"openai/gpt-image-1","name":"GPT Image 1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT 5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.1-thinking":{"id":"openai/gpt-5.1-thinking","name":"GPT 5.1 Thinking","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-11-12","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT 5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":30,"output":180}},"openai/tts-1-hd":{"id":"openai/tts-1-hd","name":"TTS-1 HD","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/o3-fast":{"id":"openai/o3-fast","name":"o3 (Fast)","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":3.5,"output":14,"cache_read":0.875}},"openai/gpt-image-1-mini":{"id":"openai/gpt-image-1-mini","name":"GPT Image 1 Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":2,"output":8,"cache_read":0.2}},"openai/gpt-5.6-terra-fast":{"id":"openai/gpt-5.6-terra-fast","name":"GPT 5.6 Terra (Fast)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":24,"cache_read":0.4,"cache_write":5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT 5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-image-2.5-sunburst":{"id":"openai/gpt-image-2.5-sunburst","name":"GPT Image 2.5 Sunburst","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-4.1-nano-fast":{"id":"openai/gpt-4.1-nano-fast","name":"GPT-4.1 nano (Fast)","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.05}},"openai/gpt-realtime-mini":{"id":"openai/gpt-realtime-mini","name":"GPT-Realtime mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-10-10","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0.6,"output":2.4,"cache_read":0.06}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-4.1-mini-fast":{"id":"openai/gpt-4.1-mini-fast","name":"GPT-4.1 mini (Fast)","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"input":1014808,"output":32768},"cost":{"input":0.7,"output":2.8,"cache_read":0.175}},"openai/gpt-realtime-whisper":{"id":"openai/gpt-realtime-whisper","name":"gpt-realtime-whisper","description":"Streaming speech-to-text model for low-latency transcript deltas from live audio","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"input":12289,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5.3-codex-fast":{"id":"openai/gpt-5.3-codex-fast","name":"GPT 5.3 Codex (Fast)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":3.5,"output":28,"cache_read":0.35}},"openai/text-embedding-3-small":{"id":"openai/text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.5}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT 5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/text-embedding-3-large":{"id":"openai/text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"input":6656,"output":1536}},"openai/gpt-image-2.5-flare":{"id":"openai/gpt-image-2.5-flare","name":"GPT Image 2.5 Flare","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT 5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.2-fast":{"id":"openai/gpt-5.2-fast","name":"GPT 5.2 (Fast)","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":3.5,"output":28,"cache_read":0.35}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-sol-fast":{"id":"openai/gpt-5.6-sol-fast","name":"GPT 5.6 Sol (Fast)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/whisper-1":{"id":"openai/whisper-1","name":"Whisper","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2022-09-21","last_updated":"2022-09-21","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"openai/gpt-4o-fast":{"id":"openai/gpt-4o-fast","name":"GPT-4o (Fast)","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"cost":{"input":4.25,"output":17,"cache_read":2.125}},"openai/gpt-realtime-2.1":{"id":"openai/gpt-realtime-2.1","name":"gpt-realtime-2.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-09-30","release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"input":96000,"output":32000},"cost":{"input":4,"output":24,"cache_read":0.4}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3 Pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":100000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT 5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":872000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code High Speed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":216144,"output":216144},"cost":{"input":0.47,"output":2,"cache_read":0.141}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k3-fast":{"id":"moonshotai/kimi-k3-fast","name":"Kimi K3 Fast","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262114,"output":262114},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"cohere/rerank-v4-pro":{"id":"cohere/rerank-v4-pro","name":"Cohere Rerank 4 Pro","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"cohere/rerank-v4-fast":{"id":"cohere/rerank-v4-fast","name":"Cohere Rerank 4 Fast","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000}},"cohere/embed-v4.0":{"id":"cohere/embed-v4.0","name":"Embed v4.0","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":1536}},"cohere/command-a":{"id":"cohere/command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/rerank-v3.5":{"id":"cohere/rerank-v3.5","name":"Cohere Rerank 3.5","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"o","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":80000},"cost":{"input":0.25,"output":0.8999999999999999}},"tencent/hy-mt2-lite":{"id":"tencent/hy-mt2-lite","name":"Tencent Hy-MT2-Lite","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.044,"output":0.177}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Tencent Hy4 Preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-plus":{"id":"tencent/hy-mt2-plus","name":"Tencent Hy-MT2-Plus","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-pro":{"id":"tencent/hy-mt2-pro","name":"Tencent Hy-MT2-Pro","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":4000},"cost":{"input":0.074,"output":0.295}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":120000},"cost":{"input":0.6,"output":2.2,"cache_read":0.12}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM 4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.2,"output":1.1,"cache_read":0.03}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.8,"output":2.55,"cache_read":0.16}},"zai/glm-5.3-fast":{"id":"zai/glm-5.3-fast","name":"GLM 5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai/glm-5.2-fast":{"id":"zai/glm-5.2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM 4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":96000},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"GLM 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":66000,"output":16000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM 4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":131100},"cost":{"input":1,"output":3.2}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202800,"output":64000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202800,"output":131100},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"zai/glm-5v-turbo":{"id":"zai/glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-4.7-flash":{"id":"zai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131000},"cost":{"input":0.07,"output":0.4}},"mistral/devstral-2":{"id":"mistral/devstral-2","name":"Devstral 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.4,"output":2}},"mistral/mistral-medium":{"id":"mistral/mistral-medium","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.4,"output":2}},"mistral/devstral-small-2":{"id":"mistral/devstral-small-2","name":"Devstral Small 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-09","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.1,"output":0.3}},"mistral/mistral-embed":{"id":"mistral/mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"mistral/mistral-large-3":{"id":"mistral/mistral-large-3","name":"Mistral Large 3","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5}},"mistral/mistral-nemo":{"id":"mistral/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-07-18","last_updated":"2024-07-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"mistral/codestral-embed":{"id":"mistral/codestral-embed","name":"Codestral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"codestral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536}},"mistral/mistral-small":{"id":"mistral/mistral-small","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2024-09-17","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4000},"cost":{"input":0.1,"output":0.3}},"mistral/ministral-14b":{"id":"mistral/ministral-14b","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-02","last_updated":"2025-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.2,"output":0.2}},"mistral/mistral-medium-3.5":{"id":"mistral/mistral-medium-3.5","name":"Mistral Medium Latest","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-05-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1.5,"output":7.5}},"mistral/codestral":{"id":"mistral/codestral","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral/ministral-8b":{"id":"mistral/ministral-8b","name":"Ministral 8B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"mistral/ministral-3b":{"id":"mistral/ministral-3b","name":"Ministral 3B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.04,"output":0.04}},"mistral/pixtral-12b":{"id":"mistral/pixtral-12b","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"perplexity/pplx-embed-v1-4b":{"id":"perplexity/pplx-embed-v1-4b","name":"Embed v1 4b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"perplexity/pplx-embed-v1-0.6b":{"id":"perplexity/pplx-embed-v1-0.6b","name":"Embed v1 0.6b","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"v0","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":0}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-02","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":8000}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":8000}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000}},"recraft/recraft-v4-pro":{"id":"recraft/recraft-v4-pro","name":"Recraft V4 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-utility":{"id":"recraft/recraft-v4.1-utility","name":"Recraft V4.1 Utility","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1-utility-pro":{"id":"recraft/recraft-v4.1-utility-pro","name":"Recraft V4.1 Utility Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v2":{"id":"recraft/recraft-v2","name":"Recraft V2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v3":{"id":"recraft/recraft-v3","name":"Recraft V3","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-30","last_updated":"2024-10","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":512,"output":0}},"recraft/recraft-v4.1-pro":{"id":"recraft/recraft-v4.1-pro","name":"Recraft V4.1 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4":{"id":"recraft/recraft-v4","name":"Recraft V4","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}},"recraft/recraft-v4.1":{"id":"recraft/recraft-v4.1","name":"Recraft V4.1","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"recraft","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-14","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0}}}},"qvac":{"id":"qvac","env":["QVAC_API_KEY"],"npm":"@qvac/ai-sdk-provider","name":"QVAC","doc":"https://www.npmjs.com/package/@qvac/ai-sdk-provider","models":{"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.5-0.8b":{"id":"qwen3.5-0.8b","name":"Qwen3.5 0.8B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen3.5-4b":{"id":"qwen3.5-4b","name":"Qwen3.5 4B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"qwen3.5-2b":{"id":"qwen3.5-2b","name":"Qwen3.5 2B","description":"Qwen instruction model for multilingual chat and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}}}},"wandb":{"id":"wandb","env":["WANDB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inference.wandb.ai/v1","name":"Weights & Biases","doc":"https://docs.wandb.ai/guides/integrations/inference/","models":{"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4-Flash is an MoE model with 1M context length great for coding, reasoning, and agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.14,"output":0.28,"cache_read":0.07}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"DeepSeek V4-Flash-0731 is an MoE model great for coding, reasoning, and agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.13,"output":0.28,"cache_read":0.07}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"A large hybrid model that supports both thinking and non-thinking modes via prompt templates.","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":161000,"output":161000},"cost":{"input":0.55,"output":1.65,"cache_read":0.55}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4-Pro-0813 is a 1.6T-parameter MoE model excelling at advanced reasoning, coding, and complex agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.31,"output":3.96,"cache_read":0.044}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4-Pro is a 1.6T-parameter MoE model with 49B active parameters excelling at advanced reasoning, coding, and complex agentic workloads.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.15,"output":2.55,"cache_read":0.2}},"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B":{"id":"nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B","name":"Nemotron 3 Ultra","description":"Nemotron 3 Ultra is a powerful MoE model designed for long-running agents across coding, deep research, and enterprise automation.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":2.75,"cache_read":0.15}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B","name":"Nemotron 3.5 Lightning","description":"Nemotron 3.5 Lightning is an MoE model built for fast, reliable agentic tasks across use cases such as financial services, cybersecurity, telecom, and retail.","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.25,"cache_read":0.05}},"OpenPipe/Qwen3-14B-Instruct":{"id":"OpenPipe/Qwen3-14B-Instruct","name":"Qwen3 14B Instruct","description":"An efficient multilingual, dense, instruction-tuned model, optimized by OpenPipe for building agents with finetuning.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.05,"output":0.22,"cache_read":0.05}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B","description":"Gemma 4 31B Dense is designed for advanced reasoning, agentic workflows, and longer context and is natively trained on 140+ languages.","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.34,"cache_read":0.1}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM 5.2","description":"GLM-5.2 is a Mixture-of-Experts language model featuring 40 billion activated parameters and a total of 744 billion parameters.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.76,"output":2.42,"cache_read":0.14}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"GLM-5.3-Flash is a natively multimodal model with 320B total parameters and 18B active parameters.","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.05}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen3-30B-A3B-Instruct-2507 is a 30.5B MoE instruction-tuned model with enhanced reasoning, coding, and long-context understanding.","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.3,"cache_read":0.1}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Qwen3.8-27B is a dense multimodal model suited for coding, research, vision, and long-running agent tasks.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":3,"cache_read":0.15}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5-35B-A3B","description":"Qwen3.5-35B-A3B is an open-weights multimodal MoE model built for efficient, high-throughput inference across chat, reasoning, and agentic tasks.","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.25,"cache_read":0.25}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen3.6-27B is a 27B dense multimodal model with 262K context built for flagship-level agentic coding.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.6,"cache_read":0.12}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Qwen3.6-35B-A3B is an MoE multimodal model with 262K context optimized for agentic coding workflows.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-15","last_updated":"2026-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.25,"cache_read":0.25}},"ibm-granite/granite-4.1-8b":{"id":"ibm-granite/granite-4.1-8b","name":"Granite 4.1 8B","description":"Granite 4.1 8B is a long-context instruct model capable of enhanced tool calling, instruction following, and chat capabilities.","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1,"cache_read":0.05}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"Granite 4.2 8B is an instruct model capable of enhanced tool calling, instruction following, and chat capabilities.","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-24","last_updated":"2026-08-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.15,"cache_read":0.05}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax M3","description":"MiniMax M3 is a multimodal MoE model with 23B active parameters optimized for coding and agentic workflows.","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.23,"output":0.96,"cache_read":0.05}},"meta-llama/Llama-3.1-8B-Instruct":{"id":"meta-llama/Llama-3.1-8B-Instruct","name":"Llama 3.1 8B","description":"Efficient conversational model optimized for responsive multilingual chatbot interactions.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.22,"output":0.22,"cache_read":0.22}},"meta-llama/Llama-3.1-70B-Instruct":{"id":"meta-llama/Llama-3.1-70B-Instruct","name":"Llama 3.1 70B","description":"Efficient conversational model optimized for responsive multilingual chatbot interactions.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.8,"output":0.8,"cache_read":0.8}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B","description":"Multilingual model excelling in conversational tasks, detailed instruction-following, and coding.","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.71,"output":0.71,"cache_read":0.71}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"gpt-oss-20b","description":"Lower latency Mixture-of-Experts model trained on OpenAI's Harmony response format with reasoning capabilities.","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.13,"cache_read":0.03}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"gpt-oss-120b","description":"Efficient Mixture-of-Experts model designed for high-reasoning, agentic and general-purpose use cases.","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.17,"cache_read":0.03}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Kimi K2.7 Code is a 1T-parameter MoE model with 32B active parameters purpose-built for long-horizon agentic coding and software engineering.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.71,"output":3.5,"cache_read":0.15}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi K2.6 is a multimodal Mixture-of-Experts language model featuring 32 billion activated parameters and a total of 1 trillion parameters.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.65,"output":3.41,"cache_read":0.15}},"JetBrains/Mellum2-12B-A2.5B-Instruct":{"id":"JetBrains/Mellum2-12B-A2.5B-Instruct","name":"Mellum2 12B A2.5B","description":"Mellum2-12B-A2.5B-Instruct is a fast MoE model with 131K context built for coding, tool use, and low-latency AI workflows.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.1,"cache_read":0.05}}}},"friendli":{"id":"friendli","env":["FRIENDLI_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.friendli.ai/serverless/v1","name":"Friendli","doc":"https://friendli.ai/docs/guides/serverless_endpoints/introduction","models":{"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.5,"output":1.5,"cache_read":0.25}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]},{"type":"budget_tokens","min":-1,"max":1048576}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}}}},"tokenrouter":{"id":"tokenrouter","env":["TOKENROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tokenrouter.com/v1","name":"TokenRouter","doc":"https://www.tokenrouter.com/docs/tokenrouter-feature-guide/","models":{"z-ai/glm-5.3-free":{"id":"z-ai/glm-5.3-free","name":"GLM-5.3 (free)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}}}},"thinkingmachines":{"id":"thinkingmachines","env":["TINKER_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://tinker.thinkingmachines.dev/services/tinker-prod/anthropic/api/v1","name":"Thinking Machines","doc":"https://tinker-docs.thinkingmachines.ai/tinker/compatible-apis/anthropic/","models":{"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":1.87,"output":4.68,"cache_read":0.374}},"thinkingmachines/Inkling:peft:262144":{"id":"thinkingmachines/Inkling:peft:262144","name":"Inkling (256K)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":3.74,"output":9.36,"cache_read":0.748}}}},"standardcompute":{"id":"standardcompute","env":["STANDARDCOMPUTE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stdcmpt.com/v1","name":"Standard Compute","doc":"https://standardcompute.com/models","models":{"standardcompute":{"id":"standardcompute","name":"Standard Compute","description":"Flat-rate smart-routing gateway: one model id, each request routed across a curated catalog of 1M-context models (DeepSeek, GLM, MiniMax, Qwen, GPT-5.6, Claude 5, Gemini 2.5, Kimi) or pinned to a user-selected model","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-08-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":24576},"cost":{"input":0,"output":0}}}},"tensorx":{"id":"tensorx","env":["TENSORX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tensorx.ai/v1","name":"TensorX","doc":"https://docs.tensorx.ai/","models":{"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.2,"cache_read":0.0375,"cache_write":0.1875}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0.06,"output":0.25,"cache_read":0.015,"cache_write":0.075}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen3 235B-A22B-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":262144},"cost":{"input":0.072,"output":0.464,"cache_read":0.018,"cache_write":0.09}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":3.5,"cache_read":0.125,"cache_write":0.625}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B-A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131072},"cost":{"input":0.21,"output":1.9,"cache_read":0.0525,"cache_write":0.2625}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.075,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.1}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.9,"cache_read":0.075,"cache_write":0.375}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek Chat V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":163840},"cost":{"input":0.2,"output":0.8,"cache_read":0.05,"cache_write":0.25}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.25,"output":0.3,"cache_read":0.06}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.3,"cache_read":0.0375,"cache_write":0.1875}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek R1-0528","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":8192},"cost":{"input":0.66,"output":2.6,"cache_read":0.165,"cache_write":0.825}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.3,"output":0.5,"cache_read":0.075,"cache_write":0.375}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.75,"output":3.5,"cache_read":0.4375,"cache_write":2.185}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.2,"cache_read":0.01,"cache_write":0.05}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1,"output":4,"cache_read":0.25,"cache_write":1.25}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.25,"output":4.5,"cache_read":0.3125}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.75}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8,"cache_read":0.125,"cache_write":0.625}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":200000},"cost":{"input":0.6,"output":2.2,"cache_read":0.15,"cache_write":0.75}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.5,"output":4.5,"cache_read":0.375}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1,"output":3.2,"cache_read":0.25,"cache_write":1.25}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.4,"output":4.4,"cache_read":0.35,"cache_write":1.75}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.3,"cache_write":1.5}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.3,"cache_write":1.5}}}},"meta":{"id":"meta","env":["META_MODEL_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.meta.ai/v1","name":"Meta","doc":"https://dev.meta.ai/docs","models":{"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}}}},"venice":{"id":"venice","env":["VENICE_API_KEY"],"npm":"venice-ai-sdk-provider","name":"Venice AI","doc":"https://docs.venice.ai","models":{"google-gemma-3-27b-it":{"id":"google-gemma-3-27b-it","name":"Google Gemma 3 27B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-11-04","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.12,"output":0.2}},"zai-org-glm-5-2":{"id":"zai-org-glm-5-2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.6,"output":18,"cache_read":0.36,"cache_write":4.5}},"deepseek-v4-flash-0731-fast":{"id":"deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731 Fast","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-08-09","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.35,"output":0.7,"cache_read":0.0875}},"qwen3-5-9b":{"id":"qwen3-5-9b","name":"Qwen 3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.1,"output":0.15}},"openai-gpt-55-pro":{"id":"openai-gpt-55-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":37.5,"output":225}},"zai-org-glm-4.7-flash":{"id":"zai-org-glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"mistral-small-3-2-24b-instruct":{"id":"mistral-small-3-2-24b-instruct","name":"Mistral Small 3.2 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-15","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.09375,"output":0.25}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.65,"output":4.95,"cache_read":0.165}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.175,"output":0.35,"cache_read":0.035}},"gemini-3-7-flash":{"id":"gemini-3-7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"venice-uncensored-1-2":{"id":"venice-uncensored-1-2","name":"Venice Uncensored 1.2","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"venice","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-01","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen 3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.45,"output":3.5}},"gemma-4-uncensored":{"id":"gemma-4-uncensored","name":"Gemma 4 Uncensored","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-04-13","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.1625,"output":0.5}},"zai-org-glm-5-1":{"id":"zai-org-glm-5-1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":80000},"cost":{"input":1.54,"output":4.84,"cache_read":0.286}},"qwen-3-6-plus":{"id":"qwen-3-6-plus","name":"Qwen 3.6 Plus Uncensored","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-06","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.625,"output":3.75,"cache_read":0.0625,"cache_write":0.78,"tiers":[{"input":2.5,"output":7.5,"cache_read":0.0625,"cache_write":0.78,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2.5,"output":7.5,"cache_read":0.0625,"cache_write":0.78}}},"openai-gpt-55":{"id":"openai-gpt-55","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":131072},"cost":{"input":6.25,"output":37.5,"cache_read":0.625,"tiers":[{"input":12.5,"output":56.25,"cache_read":1.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":12.5,"output":56.25,"cache_read":1.25}}},"claude-opus-5-fast":{"id":"claude-opus-5-fast","name":"Claude Opus 5 Fast","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-23","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"minimax-m25":{"id":"minimax-m25","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32768},"cost":{"input":0.27,"output":0.95,"cache_read":0.03}},"aion-labs-aion-3-0-mini":{"id":"aion-labs-aion-3-0-mini","name":"Aion 3.0 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.875,"output":1.75,"cache_read":0.225}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-22","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.55,"output":9.45,"cache_read":0.155,"cache_write":0.086}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-23","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"openai-gpt-52-codex":{"id":"openai-gpt-52-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08","release_date":"2025-01-15","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":272000,"output":65536},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"qwen-3-7-max":{"id":"qwen-3-7-max","name":"Qwen 3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-22","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.7,"output":8.05,"cache_read":0.27,"cache_write":3.35}},"qwen-3-8-max":{"id":"qwen-3-8-max","name":"Qwen 3.8 Max","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-22","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.5,"output":7.5,"cache_read":0.3125,"cache_write":3.125}},"openai-gpt-54":{"id":"openai-gpt-54","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":131072},"cost":{"input":3.13,"output":18.8,"cache_read":0.313}},"deepseek-v4-1-flash":{"id":"deepseek-v4-1-flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.375,"output":1.5,"cache_read":0.0075}},"openai-gpt-6-astra":{"id":"openai-gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-05","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"z-ai-glm-5-turbo":{"id":"z-ai-glm-5-turbo","name":"GLM 5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai-org-glm-4.6":{"id":"zai-org-glm-4.6","name":"GLM 4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2024-04-01","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-12-06","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":198000,"output":32768},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash 0423","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.138,"output":0.275,"cache_read":0.028}},"zai-org-glm-5":{"id":"zai-org-glm-5","name":"GLM 5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32000},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-09","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"openai-gpt-56-terra":{"id":"openai-gpt-56-terra","name":"GPT-5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"kimi-k3-fast-api":{"id":"kimi-k3-fast-api","name":"Kimi K3 Fast","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"qwen-3-8-27b":{"id":"qwen-3-8-27b","name":"Qwen 3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-17","last_updated":"2026-08-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3.2}},"openai-gpt-oss-120b":{"id":"openai-gpt-oss-120b","name":"OpenAI GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-06","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"olafangensan-glm-4.7-flash-heretic":{"id":"olafangensan-glm-4.7-flash-heretic","name":"GLM 4.7 Flash Heretic","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-02-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":24000},"cost":{"input":0.07,"output":0.4,"cache_read":0.035}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-13","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.75,"output":3.5,"cache_read":0.16}},"aion-labs-aion-3-0":{"id":"aion-labs-aion-3-0","name":"Aion 3.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":3.75,"output":7.5,"cache_read":0.9375}},"llama-3.2-3b":{"id":"llama-3.2-3b","name":"Llama 3.2 3B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-10-03","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.15,"output":0.6}},"qwen3-5-397b-a17b":{"id":"qwen3-5-397b-a17b","name":"Qwen 3.5 397B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.75,"output":4.5}},"seed-2-1-turbo":{"id":"seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-28","last_updated":"2026-07-24","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.625,"output":3.125,"cache_read":0.125}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-08-29","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":0.3,"cache_write":15}},"openai-gpt-54-pro":{"id":"openai-gpt-54-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"cost":{"input":37.5,"output":225,"tiers":[{"input":75,"output":337.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":75,"output":337.5}}},"hermes-3-llama-3.1-405b":{"id":"hermes-3-llama-3.1-405b","name":"Hermes 3 Llama 3.1 405b","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"hermes","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2025-09-25","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":3}},"venice-uncensored-role-play":{"id":"venice-uncensored-role-play","name":"Venice Role Play Uncensored","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"venice","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-02-20","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.5,"output":2}},"mercury-2-5":{"id":"mercury-2-5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-08","last_updated":"2026-09-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04999999999999999,"output":0.18749999999999994,"cache_read":0.004999999999999999}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-20","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.75,"output":3.5,"cache_read":0.16}},"openai-gpt-6-astra-pro":{"id":"openai-gpt-6-astra-pro","name":"GPT-6 Astra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-09-05","last_updated":"2026-09-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":12.5,"output":62.5,"cache_read":1.25,"cache_write":15.625,"tiers":[{"input":25,"output":93.75,"cache_read":2.5,"cache_write":31.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":25,"output":93.75,"cache_read":2.5,"cache_write":31.25}}},"openai-gpt-54-mini":{"id":"openai-gpt-54-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-27","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.9375,"output":5.625,"cache_read":0.09375}},"qwen3-6-35b-a3b":{"id":"qwen3-6-35b-a3b","name":"Qwen 3.6 35B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.1,"output":1}},"minimax-m3-preview":{"id":"minimax-m3-preview","name":"MiniMax M3 Preview","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-06-12","last_updated":"2026-06-13","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"qwen3-5-35b-a3b":{"id":"qwen3-5-35b-a3b","name":"Qwen 3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.3125,"output":1.25,"cache_read":0.15625}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"kimi-k2-5":{"id":"kimi-k2-5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2024-04","release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.56,"output":3.5,"cache_read":0.22}},"gemini-3-5-flash-lite":{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-09","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.375,"output":3.125,"cache_read":0.0375}},"google-gemma-4-26b-a4b-it":{"id":"google-gemma-4-26b-a4b-it","name":"Google Gemma 4 26B A4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.13,"output":0.4,"cache_read":0.05}},"openai-gpt-53-codex":{"id":"openai-gpt-53-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"qwen-3-8-2-4t-a95b":{"id":"qwen-3-8-2-4t-a95b","name":"Qwen 3.8 2.4T","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.3125}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3.75,"output":18.75,"cache_read":0.375}},"openai-gpt-56-sol":{"id":"openai-gpt-56-sol","name":"GPT-5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":12.5,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":18.75,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":18.75,"cache_read":0.5,"cache_write":6.25}}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":160000,"output":32768},"cost":{"input":0.33,"output":0.48,"cache_read":0.16}},"xiaomi-mimo-v2-5":{"id":"xiaomi-mimo-v2-5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-06-11","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"gemini-3-1-pro-preview":{"id":"gemini-3-1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-06-11","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2.5,"output":15,"cache_read":0.5,"cache_write":0.5,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":0.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":0.5}}},"openai-gpt-56-terra-pro":{"id":"openai-gpt-56-terra-pro","name":"GPT-5.6 Terra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32000},"cost":{"input":2.27,"output":6.8,"cache_read":0.34,"tiers":[{"input":4.53,"output":13.6,"cache_read":0.68,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":0.68}}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-10","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"openai-gpt-52":{"id":"openai-gpt-52","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-13","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":272000,"output":65536},"cost":{"input":2.19,"output":17.5,"cache_read":0.219}},"claude-opus-4-8-fast":{"id":"claude-opus-4-8-fast","name":"Claude Opus 4.8 Fast","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":12,"output":60,"cache_read":1.2,"cache_write":15}},"openai-gpt-56-luna-pro":{"id":"openai-gpt-56-luna-pro","name":"GPT-5.6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.3125,"tiers":[{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625}}},"z-ai-glm-5-3-flash":{"id":"z-ai-glm-5-3-flash","name":"GLM 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"grok-4-20":{"id":"grok-4-20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-03-12","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"google-gemma-4-31b-it":{"id":"google-gemma-4-31b-it","name":"Google Gemma 4 31B Instruct","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-03","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.12,"output":0.36,"cache_read":0.09}},"qwen3-6-27b":{"id":"qwen3-6-27b","name":"Qwen 3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.325,"output":3.25}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-18","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":1.25,"output":5.0625,"cache_read":0.2125}},"qwen-3-8-flash":{"id":"qwen-3-8-flash","name":"Qwen 3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.14,"output":0.49,"cache_read":0.014}},"qwen3-coder-480b-a35b-instruct-turbo":{"id":"qwen3-coder-480b-a35b-instruct-turbo","name":"Qwen 3 Coder 480B Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"openai-gpt-4o-2024-11-20":{"id":"openai-gpt-4o-2024-11-20","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2026-02-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":3.125,"output":12.5}},"minimax-m27":{"id":"minimax-m27","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":32768},"cost":{"input":0.375,"output":1.5,"cache_read":0.06875}},"qwen3-next-80b":{"id":"qwen3-next-80b","name":"Qwen 3 Next 80b","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.35,"output":1.9}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-02-20","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.3125,"output":0.9375,"cache_read":0.03125}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-01-15","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":198000,"output":64000},"cost":{"input":3.75,"output":18.75,"cache_read":0.375,"cache_write":4.69}},"nvidia-nemotron-3-nano-30b-a3b":{"id":"nvidia-nemotron-3-nano-30b-a3b","name":"NVIDIA Nemotron 3 Nano 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.3}},"zai-org-glm-4.7":{"id":"zai-org-glm-4.7","name":"GLM 4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.55,"output":2.65,"cache_read":0.11}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-19","last_updated":"2026-06-11","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.7,"output":3.75,"cache_read":0.07}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen 3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-29","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.75}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.1875,"output":0.75}},"z-ai-glm-5v-turbo":{"id":"z-ai-glm-5v-turbo","name":"GLM 5V Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32768},"cost":{"input":1.5,"output":5,"cache_read":0.3}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-10","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":200000},"cost":{"input":2.27,"output":6.8,"cache_read":0.57,"tiers":[{"input":4.53,"output":13.6,"cache_read":1.13,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4.53,"output":13.6,"cache_read":1.13}}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":6,"output":30,"cache_read":0.6,"cache_write":7.5}},"qwen-3-7-plus":{"id":"qwen-3-7-plus","name":"Qwen 3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":1.5,"output":6,"cache_read":0.15,"cache_write":1.875,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.5,"output":6,"cache_read":0.15,"cache_write":1.875}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.65,"output":3.301,"cache_read":0.33}},"nvidia-nemotron-3-ultra-550b-a55b":{"id":"nvidia-nemotron-3-ultra-550b-a55b","name":"NVIDIA Nemotron 3 Ultra","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.625,"output":3.125,"cache_read":0.1875}},"z-ai-glm-5-3":{"id":"z-ai-glm-5-3","name":"GLM 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.75,"output":5.5,"cache_read":0.325}},"grok-4-20-multi-agent":{"id":"grok-4-20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"release_date":"2026-03-12","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":1.42,"output":2.83,"cache_read":0.23,"tiers":[{"input":2.83,"output":5.67,"cache_read":0.45,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.83,"output":5.67,"cache_read":0.45}}},"openai-gpt-56-luna":{"id":"openai-gpt-56-luna","name":"GPT-5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.3125,"tiers":[{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.5,"output":2.25,"cache_read":0.05,"cache_write":0.625}}},"llama-3.3-70b":{"id":"llama-3.3-70b","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"release_date":"2025-04-06","last_updated":"2026-06-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.7,"output":2.8}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-29","last_updated":"2026-07-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"Qwen3 VL 235B","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2026-01-16","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.21,"output":1.9,"cache_read":0.1}},"openai-gpt-4o-mini-2024-07-18":{"id":"openai-gpt-4o-mini-2024-07-18","name":"GPT-4o Mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2026-02-28","last_updated":"2026-06-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1875,"output":0.75,"cache_read":0.09375}},"gemini-3-8-flash":{"id":"gemini-3-8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.9375,"output":4.6875,"cache_read":0.09375}},"openai-gpt-56-sol-pro":{"id":"openai-gpt-56-sol-pro","name":"GPT-5.6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":12.5,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":18.75,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":18.75,"cache_read":0.5,"cache_write":6.25}}}}},"gmicloud":{"id":"gmicloud","env":["GMICLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.gmi-serving.com/v1","name":"GMI Cloud","doc":"https://docs.gmicloud.ai/inference-engine/api-reference/llm-api-reference","models":{"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":384000},"cost":{"input":0.112,"output":0.224,"cache_read":0.022}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.392,"output":2.784,"cache_read":0.116}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":409600,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"zai-org/GLM-5.2-FP8":{"id":"zai-org/GLM-5.2-FP8","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.979,"output":3.08,"cache_read":0.182}},"zai-org/GLM-5-FP8":{"id":"zai-org/GLM-5-FP8","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.98,"output":3.08,"cache_read":0.182}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.25,"cache_write":3.125}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536},"cost":{"input":0.855,"output":3.6,"cache_read":0.144}}}},"io-net":{"id":"io-net","env":["IOINTELLIGENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.intelligence.io.solutions/api/v1","name":"IO.NET","doc":"https://io.net/docs/guides/intelligence/io-intelligence","models":{"Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar":{"id":"Intel/Qwen3-Coder-480B-A35B-Instruct-int4-mixed-ar","name":"Qwen 3 Coder 480B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":106000,"output":4096},"cost":{"input":0.22,"output":0.95,"cache_read":0.11,"cache_write":0.44}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8.75,"cache_read":1,"cache_write":4}},"mistralai/Devstral-Small-2505":{"id":"mistralai/Devstral-Small-2505","name":"Devstral Small 2505","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-05-01","last_updated":"2025-05-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.05,"output":0.22,"cache_read":0.025,"cache_write":0.1}},"mistralai/Mistral-Large-Instruct-2411":{"id":"mistralai/Mistral-Large-Instruct-2411","name":"Mistral Large Instruct 2411","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":6,"cache_read":1,"cache_write":4}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo Instruct 2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-05","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.02,"output":0.04,"cache_read":0.01,"cache_write":0.04}},"mistralai/Magistral-Small-2506":{"id":"mistralai/Magistral-Small-2506","name":"Magistral Small 2506","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0.25,"cache_write":1}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-15","last_updated":"2024-11-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.4,"output":1.75,"cache_read":0.2,"cache_write":0.8}},"Qwen/Qwen2.5-VL-32B-Instruct":{"id":"Qwen/Qwen2.5-VL-32B-Instruct","name":"Qwen 2.5 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.05,"output":0.22,"cache_read":0.025,"cache_write":0.1}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen 3 Next 80B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.1,"output":0.8,"cache_read":0.05,"cache_write":0.2}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen 3 235B Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":4096},"cost":{"input":0.11,"output":0.6,"cache_read":0.055,"cache_write":0.22}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B 128E Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":430000,"output":4096},"cost":{"input":0.15,"output":0.6,"cache_read":0.075,"cache_write":0.3}},"meta-llama/Llama-3.2-90B-Vision-Instruct":{"id":"meta-llama/Llama-3.2-90B-Vision-Instruct","name":"Llama 3.2 90B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.35,"output":0.4,"cache_read":0.175,"cache_write":0.7}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.13,"output":0.38,"cache_read":0.065,"cache_write":0.26}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":4096},"cost":{"input":0.03,"output":0.14,"cache_read":0.015,"cache_write":0.06}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.04,"output":0.4,"cache_read":0.02,"cache_write":0.08}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.55,"output":2.25,"cache_read":0.275,"cache_write":1.1}},"moonshotai/Kimi-K2-Instruct-0905":{"id":"moonshotai/Kimi-K2-Instruct-0905","name":"Kimi K2 Instruct","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2024-09-05","last_updated":"2024-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.39,"output":1.9,"cache_read":0.195,"cache_write":0.78}}}},"llmgateway":{"id":"llmgateway","env":["LLMGATEWAY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.llmgateway.io/v1","name":"DevPass (LLM Gateway)","doc":"https://llmgateway.io/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"Ministral 14B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.2}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.38,"output":1.98,"cache_read":0.19,"cache_write":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25,"cache_write":3.125}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"minimax-m2.1-lightning":{"id":"minimax-m2.1-lightning","name":"MiniMax M2.1 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.12,"output":0.48}},"gemini-pro-latest":{"id":"gemini-pro-latest","name":"Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025,"cache_write":0}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast Non-Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":2000000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"codestral-2508":{"id":"codestral-2508","name":"Codestral","description":"Mistral coding model for code completion, generation, and developer workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.3,"output":0.9}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"seed-1-8-251228":{"id":"seed-1-8-251228","name":"Seed 1.8 (251228)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"mistral-small-2506":{"id":"mistral-small-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8 27B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.2,"output":2,"cache_read":0.05}},"llama-4-scout-17b-instruct":{"id":"llama-4-scout-17b-instruct","name":"Llama 4 Scout 17B Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":2048},"cost":{"input":0.18,"output":0.59}},"qwen35-397b-a17b":{"id":"qwen35-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"qwen3-next-80b-a3b-thinking":{"id":"qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"qwen3-235b-a22b-thinking-2507":{"id":"qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking (2507)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":8192},"cost":{"input":0.3,"output":3}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.55,"output":2.2,"cache_read":0.11,"cache_write":0}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"gpt-4o-mini-transcribe":{"id":"gpt-4o-mini-transcribe","name":"GPT-4o Mini Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":1.25,"output":5}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.27,"output":1.1}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.15}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen3-coder-flash":{"id":"qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"glm-4.6v-flashx":{"id":"glm-4.6v-flashx","name":"GLM-4.6V FlashX","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.04,"output":0.4,"cache_read":0.004}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"qwen-max":{"id":"qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"ling-3.0-flash":{"id":"ling-3.0-flash","name":"InclusionAI Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-02","last_updated":"2026-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"minimax-m2":{"id":"minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.2,"output":1,"cache_read":0.03}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"minimax-m2.7-highspeed":{"id":"minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-22","last_updated":"2026-06-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":5,"output":30,"cache_read":0.5}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"qwen3-235b-a22b-fp8":{"id":"qwen3-235b-a22b-fp8","name":"Qwen3 235B A22B FP8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":8192},"cost":{"input":0.2,"output":0.8}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.08,"output":0.32,"cache_read":0.017,"cache_write":0.375}},"custom":{"id":"custom","name":"Custom Model","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.05,"cache_read":0.13}},"qwen-flash":{"id":"qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.05,"output":0.4,"cache_read":0.01,"cache_write":0.0625}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.8,"output":2.55,"cache_read":0.16,"cache_write":0}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-03","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":1050000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-2.5-flash-lite":{"id":"gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":228700,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"grok-4":{"id":"grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":3,"output":15,"cache_read":0.75}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.05,"output":0.1,"cache_read":0.01}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"seed-1-6-flash-250715":{"id":"seed-1-6-flash-250715","name":"Seed 1.6 Flash (250715)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-26","last_updated":"2025-07-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.07,"output":0.3,"cache_read":0.015}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.5,"cache_read":0.06}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.36,"output":0.87,"reasoning":8.4}},"llama-3.2-11b-instruct":{"id":"llama-3.2-11b-instruct","name":"Llama 3.2 11B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.07,"output":0.33}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"minimax-m2.5-highspeed":{"id":"minimax-m2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"llama-3.2-3b-instruct":{"id":"llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.03,"output":0.05}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"glm-4.5-x":{"id":"glm-4.5-x","name":"GLM-4.5 X","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"beta","cost":{"input":2.2,"output":8.9,"cache_read":0.45}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32766},"cost":{"input":0.04,"output":0.19,"cache_read":0.01}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.08333}},"gpt-4o-transcribe":{"id":"gpt-4o-transcribe","name":"GPT-4o Transcribe","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":16000,"output":16000},"cost":{"input":2.5,"output":10}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"qwen3-vl-plus":{"id":"qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":1.6,"reasoning":4.8,"cache_read":0.04,"cache_write":0.25}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.108,"output":0.675,"cache_read":0.06}},"qwen-coder-plus":{"id":"qwen-coder-plus","name":"Qwen Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2024-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.502,"output":1.004}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0.07,"output":0.27}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"ernie-4.5-vl-424b-a47b":{"id":"ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":123000},"cost":{"input":0.42,"output":1.25}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.08333}},"minimax-text-01":{"id":"minimax-text-01","name":"MiniMax Text 01","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.2,"output":1.1}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"glm-4.5-airx":{"id":"glm-4.5-airx","name":"GLM-4.5 AirX","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.1,"output":4.5,"cache_read":0.22}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"ministral-3b-2512":{"id":"ministral-3b-2512","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.1,"output":0.1}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen3.7-flash":{"id":"qwen3.7-flash","name":"Qwen3.7 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.0375}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k3-fast":{"id":"kimi-k3-fast","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1040384,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485}},"nemotron-3-ultra-550b":{"id":"nemotron-3-ultra-550b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"glm-5.2-fast":{"id":"glm-5.2-fast","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2.2,"output":6.5,"cache_read":0.45}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.845,"output":3.38,"cache_read":0.6,"cache_write":3.75}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"grok-4-5":{"id":"grok-4-5","name":"Grok 4.5","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.088,"output":0.25,"cache_read":0.025}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.08333}},"qwen3-vl-flash":{"id":"qwen3-vl-flash","name":"Qwen3 VL Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"qwen3-vl-30b-a3b-instruct":{"id":"qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-02","last_updated":"2025-10-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.6}},"qwen-plus":{"id":"qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.2,"reasoning":4,"cache_read":0.08,"cache_write":0.5}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":130000,"output":4096},"cost":{"input":1,"output":1}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"llama-4-maverick-17b-instruct":{"id":"llama-4-maverick-17b-instruct","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":2048},"cost":{"input":0.27,"output":0.85}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"mistral-large-latest":{"id":"mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":4,"output":12}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.05,"cache_write":0.3125}},"grok-build-0-1":{"id":"grok-build-0-1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4-3":{"id":"grok-4-3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"qwen3.6-max-preview":{"id":"qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.3,"output":7.8,"cache_read":0.13,"cache_write":1.625}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.57,"output":2.3,"cache_read":0.5}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.25,"cache_read":0.01}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":63999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":131072},"cost":{"input":0.72,"output":2.3,"cache_read":0.144,"cache_write":0}},"glm-4-32b-0414-128k":{"id":"glm-4-32b-0414-128k","name":"GLM-4 32B (0414-128k)","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.1}},"seed-1-6-250615":{"id":"seed-1-6-250615","name":"Seed 1.6 (250615)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-25","last_updated":"2025-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.405,"output":1.98,"cache_read":0.225}},"gpt-3.5-turbo":{"id":"gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5,"cache_read":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.931,"output":2.93,"cache_read":0.173,"cache_write":0}},"qwen3-vl-235b-a22b-thinking":{"id":"qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.98,"output":3.95}},"qwen3-vl-235b-a22b-instruct":{"id":"qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"llama-3.1-70b-instruct":{"id":"llama-3.1-70b-instruct","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"status":"beta","cost":{"input":0.72,"output":0.72}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct (2507)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.09,"output":0.58}},"qwen-omni-turbo":{"id":"qwen-omni-turbo","name":"Qwen-Omni Turbo","description":"Qwen omni model for text, vision, audio, and multimodal agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-01-19","last_updated":"2025-03-26","modalities":{"input":["text","image","audio","video"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32768,"output":2048},"cost":{"input":0.2,"output":0.8}},"grok-4-20-beta-0309-reasoning":{"id":"grok-4-20-beta-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"grok-4-20-beta-0309-non-reasoning":{"id":"grok-4-20-beta-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"ministral-8b-2512":{"id":"ministral-8b-2512","name":"Ministral 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"mistral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.15,"output":0.15}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32766},"cost":{"input":0.032,"output":0.14,"cache_read":0.032}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.38,"output":1.55}},"seed-1-6-250915":{"id":"seed-1-6-250915","name":"Seed 1.6 (250915)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.25,"output":2,"cache_read":0.05}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.08333}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.2}},"gpt-4":{"id":"gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"grok-4-20-non-reasoning":{"id":"grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"qwen-plus-latest":{"id":"qwen-plus-latest","name":"Qwen Plus Latest","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-01-25","last_updated":"2025-01-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":8192},"cost":{"input":0.4,"output":1.2,"cache_read":0.08,"cache_write":0.5}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1,"max":24576},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.135,"output":0.4}},"llama-3-70b-instruct":{"id":"llama-3-70b-instruct","name":"Llama 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8000},"cost":{"input":0.51,"output":0.74}},"grok-4-20-reasoning":{"id":"grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.06,"output":0.4,"cache_read":0.01,"cache_write":0}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":31999},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"auto":{"id":"auto","name":"Auto Route","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2024-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0,"output":0}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"infomaniak":{"id":"infomaniak","env":["INFOMANIAK_API_KEY","INFOMANIAK_PRODUCT_ID"],"npm":"@ai-sdk/openai-compatible","api":"https://api.infomaniak.com/2/ai/${INFOMANIAK_PRODUCT_ID}/openai/v1","name":"Infomaniak","doc":"https://www.infomaniak.com/en/hosting/ai-services/open-source-models","models":{"bge_multilingual_gemma2":{"id":"bge_multilingual_gemma2","name":"BGE Multilingual Gemma2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2024-07-25","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"input":8000,"output":3584},"cost":{"input":0.08,"output":0}},"mini_lm_l12_v2":{"id":"mini_lm_l12_v2","name":"All-MiniLM-L12-v2","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2021-08-30","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128,"input":128,"output":384},"cost":{"input":0,"output":0}},"swiss-ai/Apertus-v1.5-70B":{"id":"swiss-ai/Apertus-v1.5-70B","name":"Apertus v1.5 70B","description":"Open, ethically-sourced Swiss AI model for multilingual, multimodal chat and instruction following","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-07-24","last_updated":"2026-08-01","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":8192},"status":"beta","cost":{"input":0.87,"output":3.1}},"mistralai/Mistral-Small-4-119B-2603":{"id":"mistralai/Mistral-Small-4-119B-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.25,"output":0.93}},"mistralai/Ministral-3-14B-Instruct-2512":{"id":"mistralai/Ministral-3-14B-Instruct-2512","name":"Ministral 3 14B Instruct","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":25600},"status":"beta","cost":{"input":0.37,"output":0.5}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B-FP8","name":"Nemotron 3 Nano 30B A3B FP8","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"input":1000000,"output":262144},"status":"beta","cost":{"input":0.06,"output":0.25}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":100000,"input":100000,"output":32768},"cost":{"input":0.25,"output":0.5}},"Qwen/Qwen3.5-122B-A10B-FP8":{"id":"Qwen/Qwen3.5-122B-A10B-FP8","name":"Qwen3.5 122B-A10B FP8","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65536},"cost":{"input":0.5,"output":3.97}},"Qwen/Qwen3.5-397B-A17B-FP8":{"id":"Qwen/Qwen3.5-397B-A17B-FP8","name":"Qwen3.5 397B-A17B FP8","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"input":200000,"output":65536},"status":"beta","cost":{"input":0.99,"output":4.46}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-08-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"status":"beta","cost":{"input":0.74,"output":3.72}}}},"inception":{"id":"inception","env":["INCEPTION_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inceptionlabs.ai/v1/","name":"Inception","doc":"https://platform.inceptionlabs.ai/docs","models":{"mercury-2.5":{"id":"mercury-2.5","name":"Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-11-01","release_date":"2026-09-08","last_updated":"2026-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"mercury-edit-2":{"id":"mercury-edit-2","name":"Mercury Edit 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"mercury-2":{"id":"mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}}}},"lilac":{"id":"lilac","env":["LILAC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.getlilac.com/v1","name":"Lilac","doc":"https://docs.getlilac.com/inference/models","models":{"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":262100},"cost":{"input":0.11,"output":0.35}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":524288},"cost":{"input":0.9,"output":3,"cache_read":0.27}},"minimaxai/minimax-m3":{"id":"minimaxai/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax-m3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.28,"output":1.1,"cache_read":0.05}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":3.5,"cache_read":0.2}}}},"fastrouter":{"id":"fastrouter","env":["FASTROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://go.fastrouter.ai/api/v1","name":"FastRouter","doc":"https://fastrouter.ai/models","models":{"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen3 Coder","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.3,"output":1.2}},"deepseek-ai/deepseek-r1-distill-llama-70b":{"id":"deepseek-ai/deepseek-r1-distill-llama-70b","name":"DeepSeek R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-10","release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.03,"output":0.14}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"google/veo3.1-fast":{"id":"google/veo3.1-fast","name":"Veo 3.1 Fast","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/veo3.1":{"id":"google/veo3.1","name":"Veo 3.1","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12}},"google/veo3.1-lite":{"id":"google/veo3.1-lite","name":"Veo 3.1 Lite","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"veo","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-01","last_updated":"2026-05-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":400000,"output":0}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"google/imagen-4.0-ultra":{"id":"google/imagen-4.0-ultra","name":"Imagen 4 Ultra","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/imagen-4.0-fast":{"id":"google/imagen-4.0-fast","name":"Imagen 4 Fast","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"imagen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":480,"output":0}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.31}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.0375}},"bytedance/seedance-2":{"id":"bytedance/seedance-2","name":"Seedance 2","description":"Video model for prompt-guided generation, editing, and motion workflows","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":false,"limit":{"context":4096,"output":0}},"wanx/wan-v2-6":{"id":"wanx/wan-v2-6","name":"Wan 2.6","description":"Video model for prompt-guided generation, editing, and motion workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image"],"output":["video"]},"open_weights":true,"limit":{"context":400000,"output":0}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48}},"leonardo-ai/lucid-realism":{"id":"leonardo-ai/lucid-realism","name":"Lucid Realism","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"lucid","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0}},"leonardo-ai/lucid-origin":{"id":"leonardo-ai/lucid-origin","name":"Lucid Origin","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"lucid","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":4096,"output":0}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5}},"x-ai/grok-4":{"id":"x-ai/grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.75,"cache_write":15}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-realtime-1.5":{"id":"openai/gpt-realtime-1.5","name":"GPT Realtime 1.5","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text","audio","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":32000,"output":4096},"cost":{"input":4,"output":16}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.05,"output":0.2}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5}},"openai/gpt-image-2":{"id":"openai/gpt-image-2","name":"GPT Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":128000,"output":0}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.55,"output":2.2}},"sarvam/sarvam-105b":{"id":"sarvam/sarvam-105b","name":"Sarvam 105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.04,"output":0.16}},"sarvam/sarvam-30b":{"id":"sarvam/sarvam-30b","name":"Sarvam 30B","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.02,"output":0.1}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.95,"output":3.15}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.05,"output":3.5}}}},"cloudflare-ai-gateway":{"id":"cloudflare-ai-gateway","env":["CLOUDFLARE_API_TOKEN","CLOUDFLARE_ACCOUNT_ID","CLOUDFLARE_GATEWAY_ID"],"npm":"ai-gateway-provider","name":"Cloudflare AI Gateway","doc":"https://developers.cloudflare.com/ai-gateway/","models":{"alibaba/qwen3.7-max":{"id":"alibaba/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":3.75,"cache_read":0.25}},"alibaba/qwen3.5-397b-a17b":{"id":"alibaba/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"alibaba/qwen3-max":{"id":"alibaba/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6}},"alibaba/qwen3.8-max":{"id":"alibaba/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"alibaba/qwen3.7-plus":{"id":"alibaba/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.32,"output":1.28,"cache_read":0.064}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":10,"cache_read":0.25,"cache_write":3.125}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":5,"cache_read":0.625}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":5,"output":30}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6,"cache_read":0.2}}}},"github-copilot":{"id":"github-copilot","env":["GITHUB_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.githubcopilot.com","name":"GitHub Copilot","doc":"https://docs.github.com/en/copilot","models":{"claude-opus-4.8":{"id":"claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":64000},"experimental":{"modes":{"fast":{"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"claude-opus-4.7":{"id":"claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":32000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":224000,"output":32000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"claude-sonnet-4.6":{"id":"claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":168000,"output":32000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":256,"max":32000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"mai-code-1-flash-picker":{"id":"mai-code-1-flash-picker","name":"MAI-Code-1-Flash","description":"Microsoft coding model built for fast, efficient assistance in everyday developer workflows","family":"mai","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-06-02","last_updated":"2026-06-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":128000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"claude-haiku-4.5":{"id":"claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024,"max":32000}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":136000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]},{"type":"budget_tokens","min":256,"max":24000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":128000,"output":64000},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"mai-code-1.1-flash":{"id":"mai-code-1.1-flash","name":"MAI-Code-1.1-Flash","description":"Microsoft coding model with native vision support, optimized for fast and efficient software development","family":"mai","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":128000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"input":372000,"output":128000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":264000,"input":128000,"output":64000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":936000,"output":64000},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-fable-5.1":{"id":"claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"zhipuai":{"id":"zhipuai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://open.bigmodel.cn/api/paas/v4","name":"Zhipu AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":5,"output":22,"cache_read":1.2,"cache_write":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}}}},"jalapeno":{"id":"jalapeno","env":["JALAPENO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.jalapeno-cloud.ai/v1","name":"Jalapeno Cloud","doc":"https://www.jalapeno-cloud.ai/docs/","models":{"Qwen3.5-27B":{"id":"Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2.4}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.38,"output":4.4}},"Qwen3.5-122B-A10B":{"id":"Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":3.2}},"Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":129024,"output":32768},"cost":{"input":0.3,"output":1.5}},"Kimi-K2.5":{"id":"Kimi-K2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":180224},"cost":{"input":0.6,"output":3}},"Kimi-K2.7-Code":{"id":"Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":271360,"output":262144},"cost":{"input":0.95,"output":4}},"Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":129024,"output":32768},"cost":{"input":0.15,"output":1.5}},"Qwen3.5-397B-A17B":{"id":"Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3.6}},"Qwen3.5-35B-A3B":{"id":"Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.25,"output":2}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2}},"Hy3":{"id":"Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58}},"Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen3-VL-235B-A22B-Thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.98,"output":3.95}},"Qwen3-Next-80B-A3B-Thinking":{"id":"Qwen3-Next-80B-A3B-Thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.5}},"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.6,"output":3.38}}}},"perplexity-agent":{"id":"perplexity-agent","env":["PERPLEXITY_API_KEY"],"npm":"@ai-sdk/openai","api":"https://api.perplexity.ai/v1","name":"Perplexity Agent","doc":"https://docs.perplexity.ai/docs/agent-api/models","models":{"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32000},"cost":{"input":0.25,"output":2.5}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"tiers":[{"input":0.5,"output":3,"cache_read":0.05,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.5,"output":3,"cache_read":0.05}}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-03-20","last_updated":"2025-06-05","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"moonshot-ai/kimi-k2.7-code":{"id":"moonshot-ai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-07-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot-ai/kimi-k3":{"id":"moonshot-ai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-30","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.13,"output":0.26,"cache_read":0.028}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"xai/grok-4-1-fast-non-reasoning":{"id":"xai/grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.25,"output":2.5,"cache_read":0.0625}}}},"fireworks-ai":{"id":"fireworks-ai","env":["FIREWORKS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.fireworks.ai/inference/v1/","name":"Fireworks AI","doc":"https://fireworks.ai/docs/","models":{"accounts/fireworks/routers/kimi-k3-fast":{"id":"accounts/fireworks/routers/kimi-k3-fast","name":"Kimi K3 Fast","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":4.5,"output":22.5,"cache_read":0.45}},"accounts/fireworks/routers/glm-5p3-fast":{"id":"accounts/fireworks/routers/glm-5p3-fast","name":"GLM 5.3 Fast","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-09-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048572,"output":262144},"cost":{"input":2.1,"output":6.6,"cache_read":0.39}},"accounts/fireworks/routers/glm-5p2-fast":{"id":"accounts/fireworks/routers/glm-5p2-fast","name":"GLM 5.2 Fast","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-26","last_updated":"2026-06-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":131072},"cost":{"input":2.1,"output":6.6,"cache_read":0.21}},"accounts/fireworks/models/qwen3p7-plus":{"id":"accounts/fireworks/models/qwen3p7-plus","name":"Qwen 3.7 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.08}},"accounts/fireworks/models/deepseek-v4-flash-vision-exp":{"id":"accounts/fireworks/models/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/deepseek-v4-pro-0813":{"id":"accounts/fireworks/models/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"accounts/fireworks/models/deepseek-v4-flash-0731":{"id":"accounts/fireworks/models/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/minimax-m3":{"id":"accounts/fireworks/models/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"accounts/fireworks/models/deepseek-v4p1-flash":{"id":"accounts/fireworks/models/deepseek-v4p1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"accounts/fireworks/models/kimi-k2p6":{"id":"accounts/fireworks/models/kimi-k2p6","name":"Kimi K2.6","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"accounts/fireworks/models/nemotron-3-ultra-nvfp4":{"id":"accounts/fireworks/models/nemotron-3-ultra-nvfp4","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.6,"output":2.4,"cache_read":0.119}},"accounts/fireworks/models/kimi-k3":{"id":"accounts/fireworks/models/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"accounts/fireworks/models/glm-5p3":{"id":"accounts/fireworks/models/glm-5p3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"accounts/fireworks/models/kimi-k2p7-code":{"id":"accounts/fireworks/models/kimi-k2p7-code","name":"Kimi K2.7 Code","description":"Kimi coding model for software agents, refactors, and repository reasoning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"accounts/fireworks/models/glm-5p3-flash":{"id":"accounts/fireworks/models/glm-5p3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-07","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048573,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"accounts/fireworks/models/qwen3p8-2p4t-a95b":{"id":"accounts/fireworks/models/qwen3p8-2p4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/models/muse-glimmer-30b":{"id":"accounts/fireworks/models/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"accounts/fireworks/models/inkling":{"id":"accounts/fireworks/models/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"accounts/fireworks/models/gpt-oss-120b":{"id":"accounts/fireworks/models/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"accounts/fireworks/models/mistral-large-3-fp8":{"id":"accounts/fireworks/models/mistral-large-3-fp8","name":"Mistral Large 3 675B Instruct 2512","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"accounts/fireworks/models/glm-5p2":{"id":"accounts/fireworks/models/glm-5p2","name":"GLM 5.2","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"accounts/fireworks/models/qwen3p8-max":{"id":"accounts/fireworks/models/qwen3p8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b":{"id":"accounts/fireworks/models/nemotron-lightning-3p5-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.01}}}},"opper":{"id":"opper","env":["OPPER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.opper.ai/v3/compat","name":"Opper","doc":"https://opper.ai/models","models":{"minimax/m3":{"id":"minimax/m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"tier":{"type":"context","size":524288}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24}}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"gemini/gemini-3.1-pro-preview":{"id":"gemini/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gemini/gemini-3.5-flash":{"id":"gemini/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"gemini/gemini-3.5-flash-lite":{"id":"gemini/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gemini/gemini-3-flash-preview":{"id":"gemini/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.3-chat-latest":{"id":"openai/gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/mistral-small-2603":{"id":"mistral/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"vertexai/gemini-3.7-flash-eu":{"id":"vertexai/gemini-3.7-flash-eu","name":"Gemini 3.7 Flash (EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"vertexai/gemini-3.7-flash":{"id":"vertexai/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}}}},"stackit":{"id":"stackit","env":["STACKIT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.openai-compat.model-serving.eu01.onstackit.cloud/v1","name":"STACKIT","doc":"https://docs.stackit.cloud/products/data-and-ai/ai-model-serving/basics/available-shared-models","models":{"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-05-17","last_updated":"2025-05-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":37000,"output":4096},"cost":{"input":0.53,"output":0.76}},"Qwen/Qwen3-VL-Embedding-8B":{"id":"Qwen/Qwen3-VL-Embedding-8B","name":"Qwen3-VL Embedding 8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.09,"output":0.09}},"Qwen/Qwen3-VL-235B-A22B-Instruct-FP8":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct-FP8","name":"Qwen3-VL 235B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-11-01","last_updated":"2024-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":218000,"output":16384},"cost":{"input":1.76,"output":2.05}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.53,"output":0.76}},"intfloat/e5-mistral-7b-instruct":{"id":"intfloat/e5-mistral-7b-instruct","name":"E5 Mistral 7B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.02,"output":0.02}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.18,"output":0.29}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":8192},"cost":{"input":0.53,"output":0.76}},"cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic":{"id":"cortecs/Llama-3.3-70B-Instruct-FP8-Dynamic","name":"Llama 3.3 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.53,"output":0.76}}}},"crof":{"id":"crof","env":["CROF_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://crof.ai/v1","name":"CrofAI","doc":"https://crof.ai/docs","models":{"greg-2-super":{"id":"greg-2-super","name":"Greg 2 Super","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":1.5,"output":5,"cache_read":0.25}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.08,"output":0.2,"cache_read":0.007}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro (0813)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":0.8,"cache_read":0.01}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash (New)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.08,"output":0.1,"cache_read":0.003}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-13","last_updated":"2026-03-13","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.04,"output":0.15,"cache_read":0.008}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.5,"cache_read":0.03}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.5,"output":1.99,"cache_read":0.05}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.3,"output":1.05,"cache_read":0.05}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.12,"output":0.21,"cache_read":0.003}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.55,"output":2.25,"cache_read":0.05}},"greg-1-mini":{"id":"greg-1-mini","name":"Greg 1 Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":0.07,"output":0.15,"cache_read":0.01}},"greg-2-ultra":{"id":"greg-2-ultra","name":"Greg 2 Ultra","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":3,"output":10,"cache_read":0.5}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":1.75,"cache_read":0.07}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.2,"output":1.5,"cache_read":0.04}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":2,"output":8,"cache_read":0.25}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":0.18,"output":0.35,"cache_read":0.04}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM 5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.07,"output":0.22,"cache_read":0.01}},"greg-rp":{"id":"greg-rp","name":"Greg (Roleplay)","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":229376,"output":229376},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.45,"output":2.15,"cache_read":0.08,"cache_write":0}},"kimi-k3-eco":{"id":"kimi-k3-eco","name":"Kimi K3 Eco","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":1,"output":4,"cache_read":0.1}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.35,"output":0.8,"cache_read":0.003}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.4,"output":1.4,"cache_read":0.06}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai-compatible"},"cost":{"input":0.4,"output":0.8,"cache_read":0.003,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}}}},"crusoe":{"id":"crusoe","env":["CRUSOE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.inference.crusoecloud.com/v1","name":"Crusoe","doc":"https://docs.crusoecloud.com/managed-inference/overview","models":{"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.5,"output":1.5,"cache_read":0.25}},"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B":{"id":"nvidia/NVIDIA-Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B":{"id":"nvidia/Nemotron-3-Nano-Omni-Reasoning-30B-A3B","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.3,"output":1.83,"cache_read":0.3,"input_audio":0.5}},"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B":{"id":"nvidia/NVIDIA-Nemotron-3-Super-120B-A12B","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.4,"cache_read":0.15}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.14,"output":0.4,"cache_read":0.14}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.8,"cache_read":0.11}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.25,"output":0.75,"cache_read":0.13}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2,"cache_read":0.05}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.7,"output":3.5,"cache_read":0.35}},"zai/GLM-5.1":{"id":"zai/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4.4,"cache_read":0.25}},"zai/GLM-5.2":{"id":"zai/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}}}},"empiriolabs":{"id":"empiriolabs","env":["EMPIRIOLABS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.empiriolabs.ai/v1","name":"EmpirioLabs AI","doc":"https://docs.empiriolabs.ai","models":{"glm-5-1":{"id":"glm-5-1","name":"GLM 5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":38912}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":128000},"cost":{"input":0.825,"output":3.301,"cache_read":0.165,"tiers":[{"input":1.1,"output":3.851,"cache_read":0.22,"tier":{"type":"context","size":32000}}]}},"qwen3-5-9b":{"id":"qwen3-5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":0.13,"cache_read":0.045}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"kimi-k2-7-code-highspeed":{"id":"kimi-k2-7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":1.9,"output":8,"cache_read":1.9}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.424,"output":1.272,"cache_read":0.424}},"mistral-small-4":{"id":"mistral-small-4","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0.15,"output":0.6,"cache_read":0.15}},"minimax-m2-7-highspeed":{"id":"minimax-m2-7-highspeed","name":"MiniMax M2.7 Highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"qwen3-8-max-0902":{"id":"qwen3-8-max-0902","name":"Qwen3.8 Max 0902","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":2}},"seed-2-0-pro":{"id":"seed-2-0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.63,"output":3.79,"cache_read":0.63,"tiers":[{"input":1.26,"output":7.58,"cache_read":1.26,"tier":{"type":"context","size":128000}}]}},"qwen3-7-plus":{"id":"qwen3-7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":256000}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.4,"tiers":[{"input":1.2,"output":4.8,"cache_read":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":1.2}}},"qwen3-6-flash":{"id":"qwen3-6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":64000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.25,"tiers":[{"input":1,"output":4,"cache_read":1,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1,"output":4,"cache_read":1}}},"qwen3-5-27b":{"id":"qwen3-5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.086,"output":0.688,"cache_read":0.086,"tiers":[{"input":0.258,"output":2.064,"cache_read":0.258,"tier":{"type":"context","size":128000}}]}},"glm-4-6v-flash":{"id":"glm-4-6v-flash","name":"GLM 4.6V Flash","description":"Lightweight GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0,"output":0}},"seed-2-0-mini":{"id":"seed-2-0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.12,"output":0.5,"cache_read":0.12,"tiers":[{"input":0.24,"output":1,"cache_read":0.24,"tier":{"type":"context","size":128000}}]}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":524288},"cost":{"input":0.225,"output":0.9,"cache_read":0.045,"tiers":[{"input":0.45,"output":1.8,"cache_read":0.09,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.45,"output":1.8,"cache_read":0.09}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"qwen3-5-4b":{"id":"qwen3-5-4b","name":"Qwen3.5 4B","description":"Qwen3.5 4B is a low-cost multimodal reasoning model with 256K context, image and video input, function tools, and structured output.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-02","last_updated":"2026-03-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.04,"output":0.07,"cache_read":0.02}},"glm-5-3":{"id":"glm-5-3","name":"GLM 5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"kimi-k2-7-code":{"id":"kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"fugu-ultra-v1-1":{"id":"fugu-ultra-v1-1","name":"Fugu Ultra v1.1","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"step-3-5-flash-2603":{"id":"step-3-5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"qwen3-5-397b-a17b":{"id":"qwen3-5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":1.032,"cache_read":0.172,"tiers":[{"input":0.43,"output":2.58,"cache_read":0.43,"tier":{"type":"context","size":128000}}]}},"seed-2-1-turbo":{"id":"seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":0.63,"output":3.13,"cache_read":0.63}},"deepseek-v3-2":{"id":"deepseek-v3-2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.57,"output":1.71,"cache_read":0.57}},"glm-4-7-flash":{"id":"glm-4-7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0}},"kimi-k2-6":{"id":"kimi-k2-6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":81920}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16000},"cost":{"input":0.8939,"output":3.7131,"cache_read":0.1788}},"gemma-4-26b-a4b":{"id":"gemma-4-26b-a4b","name":"Gemma 4 26B-A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.29,"cache_read":0.025}},"qwen3-6-35b-a3b":{"id":"qwen3-6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.07,"output":0.42,"cache_read":0.035}},"qwen3-5-35b-a3b":{"id":"qwen3-5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.057,"output":0.459,"cache_read":0.057,"tiers":[{"input":0.229,"output":1.835,"cache_read":0.229,"tier":{"type":"context","size":128000}}]}},"muse-spark-1-2":{"id":"muse-spark-1-2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"qwen3-6-plus":{"id":"qwen3-6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.5,"tiers":[{"input":2,"output":6,"cache_read":2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":2}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":3,"output":15,"cache_read":3}},"qwen3-5-122b-a10b":{"id":"qwen3-5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.115,"output":0.917,"cache_read":0.115,"tiers":[{"input":0.287,"output":2.294,"cache_read":0.287,"tier":{"type":"context","size":128000}}]}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.08,"output":5.52,"cache_read":1.08,"tiers":[{"input":2.16,"output":11.04,"cache_read":2.16,"tier":{"type":"context","size":32000}},{"input":2.7,"output":13.8,"cache_read":2.7,"tier":{"type":"context","size":128000}}]}},"glm-4-5-flash":{"id":"glm-4-5-flash","name":"GLM 4.5 Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":98304},"cost":{"input":0,"output":0}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.075}},"step-3-5-flash":{"id":"step-3-5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.1,"output":0.3,"cache_read":0.02}},"muse-glimmer-30b":{"id":"muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":0.8,"cache_read":0.05}},"qwen3-6-27b":{"id":"qwen3-6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":80000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.412564,"output":2.475384,"cache_read":0.412564}},"qwen3-8-27b":{"id":"qwen3-8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.17,"output":0.5,"cache_read":0.08}},"muse-spark-1-1":{"id":"muse-spark-1-1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"glm-5-2":{"id":"glm-5-2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"seed-2-0-code":{"id":"seed-2-0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.4,"output":2.4,"cache_read":0.4,"tiers":[{"input":0.8,"output":4.8,"cache_read":0.8,"tier":{"type":"context","size":128000}}]}},"qwen3-7-max":{"id":"qwen3-7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":64000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":2.5}},"mimo-v2-5":{"id":"mimo-v2-5","name":"MiMo V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.7,"output":1.4,"cache_read":0.014}},"qwen3-5-flash":{"id":"qwen3-5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.09,"output":0.368,"cache_read":0.09}},"seed-2-0-lite":{"id":"seed-2-0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.31,"output":2.5,"cache_read":0.31,"tiers":[{"input":0.62,"output":5,"cache_read":0.62,"tier":{"type":"context","size":128000}}]}},"muse-spark-1-3":{"id":"muse-spark-1-3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":1}},"mimo-v2-5-pro":{"id":"mimo-v2-5-pro","name":"MiMo V2.5 Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":2.175,"output":4.35,"cache_read":0.018}},"step-3-7-flash":{"id":"step-3-7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":131072},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.65,"output":3.3,"cache_read":1.65}},"fugu-ultra-v1-0":{"id":"fugu-ultra-v1-0","name":"Fugu Ultra v1.0","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":7.5,"output":45,"cache_read":1.5,"tiers":[{"input":15,"output":67.5,"cache_read":3,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":15,"output":67.5,"cache_read":3}}},"qwen3-8-max":{"id":"qwen3-8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":2}},"minimax-m2-7":{"id":"minimax-m2-7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.03}},"qwen3-5-plus":{"id":"qwen3-5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.36,"output":2.21,"cache_read":0.36,"tiers":[{"input":1.08,"output":6.62,"cache_read":1.08,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.08,"output":6.62,"cache_read":1.08}}},"qwen3-7-flash":{"id":"qwen3-7-flash","name":"Qwen3.7 Flash","description":"Lightweight multimodal Qwen model for high-throughput text, image, and video tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":131072}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"tier":{"type":"context","size":256000}}]}},"qwen3-8-flash":{"id":"qwen3-8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.16,"output":0.47,"cache_read":0.16}},"qwen3-6-max-preview":{"id":"qwen3-6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]},{"type":"budget_tokens","min":1,"max":393216}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.31,"output":7.88,"cache_read":1.31,"tiers":[{"input":1.97,"output":11.82,"cache_read":1.97,"tier":{"type":"context","size":128000}}]}}}},"klokintegration":{"id":"klokintegration","env":["KLOKINTEGRATION_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api-gw.klok.ipaas.se/proxy/kloker-key/v1","name":"klokintegration.se","doc":"https://klokintegration.se/docs/ai-api","models":{"Kloker-Integration-Developer":{"id":"Kloker-Integration-Developer","name":"Kloker Integration Developer","description":"Knows the customer integration environment and Klok best practices. Opinionated about implementation. The gateway runs ecosystem lookup tools server-side and appends a system-prompt injection. Client system prompts and OpenAI tool calls are preserved.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}},"Kloker-Integration-Architect":{"id":"Kloker-Integration-Architect","name":"Kloker Integration Architect","description":"Knows the customer integration environment and Klok best practices. Opinionated about structure. The gateway runs ecosystem lookup tools server-side and appends a system-prompt injection (data contracts, CloudEvents, event-driven flows). Client system prompts and OpenAI tool calls are preserved.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}},"Kloker":{"id":"Kloker","name":"Kloker","description":"Cheap general model with a clean context. Nothing from the customer environment is packed in. It tracks the current best open source model. The Klok team verifies it and upgrades it periodically.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-29","last_updated":"2026-08-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":50000},"status":"beta","cost":{"input":0.23,"output":1.16}}}},"privatemode-ai":{"id":"privatemode-ai","env":["PRIVATEMODE_API_KEY","PRIVATEMODE_ENDPOINT"],"npm":"@ai-sdk/openai-compatible","api":"http://localhost:8080/v1","name":"Privatemode AI","doc":"https://docs.privatemode.ai/api/overview","models":{"kimi-latest":{"id":"kimi-latest","name":"Kimi (latest)","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":262144},"cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"voxtral-mini-3b":{"id":"voxtral-mini-3b","name":"Voxtral Mini 3B","description":"Speech-to-text model for audio transcription, translation, and audio understanding","family":"voxtral","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07","last_updated":"2025-07","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.00462,"output":0}},"qwen3-embedding-4b":{"id":"qwen3-embedding-4b","name":"Qwen3-Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-06","release_date":"2025-06-06","last_updated":"2025-06-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2560},"cost":{"input":0.1502,"output":0}},"whisper-large-v3":{"id":"whisper-large-v3","name":"Whisper large-v3","description":"Open Whisper checkpoint for robust multilingual transcription and captioning","family":"whisper","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2024-10-01","last_updated":"2024-10-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":4096},"cost":{"input":0.01618,"output":0}},"glm-latest":{"id":"glm-latest","name":"GLM (latest)","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.4969,"output":1.9644,"cache_read":0.0462}},"deepseek-ocr-2":{"id":"deepseek-ocr-2","name":"DeepSeek OCR 2","description":"High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes","attachment":true,"reasoning":false,"tool_call":false,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"status":"beta","cost":{"input":0.8897,"output":1.4675,"cache_read":0.0924}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":131072},"status":"beta","cost":{"input":1.791,"output":8.9436,"cache_read":0.1733}}}},"minimax-coding-plan":{"id":"minimax-coding-plan","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimax.io/anthropic/v1","name":"MiniMax Token Plan (minimax.io)","doc":"https://platform.minimax.io/docs/token-plan/intro","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"inferx":{"id":"inferx","env":["INFERX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://model.inferx.net/endpoints/v1","name":"InferX","doc":"https://model.inferx.net/endpoints","models":{"gemma-4-31B-it-fp8":{"id":"gemma-4-31B-it-fp8","name":"Gemma 4 31B IT FP8","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"Qwen3-Coder-Next-FP8-no-thinking":{"id":"Qwen3-Coder-Next-FP8-no-thinking","name":"Qwen3-Coder-Next-FP8-no-thinking","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":260000,"output":65536},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"deepseek-v4-flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":100000},"cost":{"input":0,"output":0}},"Devstral-2-123B-Instruct-2512-int4-AutoRound":{"id":"Devstral-2-123B-Instruct-2512-int4-AutoRound","name":"Devstral-2-123B-Instruct-2512-int4-AutoRound","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0,"output":0}},"Agents-A1":{"id":"Agents-A1","name":"Agents-A1","description":"35B MoE agentic model built for long-horizon search, engineering, and scientific reasoning tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"release_date":"2026-06-26","last_updated":"2026-06-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":100000},"cost":{"input":0,"output":0}},"Ornith-1.0-35B-FP8":{"id":"Ornith-1.0-35B-FP8","name":"Ornith-1.0-35B-FP8","description":"Large coding-reasoning model for agentic software tasks and RL search","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-25","last_updated":"2026-06-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":100000},"cost":{"input":0,"output":0}},"Qwen3.6-35B-A3B-FP8":{"id":"Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0,"output":0}},"Qwen3.6-27B-FP8":{"id":"Qwen3.6-27B-FP8","name":"Qwen3.6 27B FP8","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"Qwen3.6-35B-A3B-fp8-no-thinking":{"id":"Qwen3.6-35B-A3B-fp8-no-thinking","name":"Qwen3.6-35B-A3B-fp8-no-thinking","description":"Qwen3.6-35B-A3B-fp8 disable thinking","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65536},"cost":{"input":0,"output":0}},"Qwen3-Coder-Next-FP8":{"id":"Qwen3-Coder-Next-FP8","name":"Qwen3 Coder Next FP8","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256144,"output":65536},"cost":{"input":0,"output":0}},"Qwen3-Embedding-8B":{"id":"Qwen3-Embedding-8B","name":"Qwen3-Embedding-8B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":0},"cost":{"input":0,"output":0}},"mimo-v25":{"id":"mimo-v25","name":"mimo-v25","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":100000},"cost":{"input":0,"output":0}}}},"umans-ai-coding-plan":{"id":"umans-ai-coding-plan","env":["UMANS_AI_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.code.umans.ai/v1","name":"Umans AI Coding Plan","doc":"https://app.umans.ai/offers/code/docs","models":{"umans-qwen3.6-35b-a3b":{"id":"umans-qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-glm-5.2":{"id":"umans-glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":405504,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-kimi-k3":{"id":"umans-kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-kimi-k2.7":{"id":"umans-kimi-k2.7","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-deepseek-v4-flash-0731":{"id":"umans-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-coder":{"id":"umans-coder","name":"Umans Coder","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-flash":{"id":"umans-flash","name":"Umans Flash","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"umans-deepseek-v4-pro-0813":{"id":"umans-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"databricks":{"id":"databricks","env":["DATABRICKS_HOST","DATABRICKS_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://${DATABRICKS_HOST}/ai-gateway/mlflow/v1","name":"Databricks","doc":"https://docs.databricks.com/aws/en/machine-learning/foundation-models/","models":{"databricks-claude-opus-4-5":{"id":"databricks-claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-claude-sonnet-4-6":{"id":"databricks-claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gemini-3-pro":{"id":"databricks-gemini-3-pro","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"databricks-kimi-k2-7-code":{"id":"databricks-kimi-k2-7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"databricks-gpt-5-6-luna":{"id":"databricks-gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"tiers":[{"input":2,"output":9,"cache_read":0.2,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2}}},"databricks-claude-opus-4-1":{"id":"databricks-claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"databricks-gpt-5-mini":{"id":"databricks-gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"databricks-gemini-2-5-flash":{"id":"databricks-gemini-2-5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"databricks-claude-haiku-4-5":{"id":"databricks-claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"databricks-claude-sonnet-4-5":{"id":"databricks-claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-4":{"id":"databricks-gpt-5-4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"databricks-gpt-5-6-sol":{"id":"databricks-gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"databricks-glm-5-2":{"id":"databricks-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"databricks-gpt-5-4-nano":{"id":"databricks-gpt-5-4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"databricks-gpt-5-5":{"id":"databricks-gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"databricks-gemini-3-1-flash-lite":{"id":"databricks-gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"databricks-gemini-3-flash":{"id":"databricks-gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"databricks-claude-opus-4-7":{"id":"databricks-claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-claude-opus-4-6":{"id":"databricks-claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"databricks-claude-sonnet-4":{"id":"databricks-claude-sonnet-4","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"databricks-gpt-5-1":{"id":"databricks-gpt-5-1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"databricks-gpt-oss-20b":{"id":"databricks-gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2}},"databricks-gpt-5-4-mini":{"id":"databricks-gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"databricks-gemini-3-1-pro":{"id":"databricks-gemini-3-1-pro","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"databricks-gpt-5-6-terra":{"id":"databricks-gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"databricks-gemini-2-5-pro":{"id":"databricks-gemini-2-5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"databricks-gpt-5":{"id":"databricks-gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"databricks-gpt-5-nano":{"id":"databricks-gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"databricks-gpt-oss-120b":{"id":"databricks-gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.072,"output":0.28}},"databricks-gpt-5-2":{"id":"databricks-gpt-5-2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}}}},"modal":{"id":"modal","env":["MODAL_PROXY_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.us-west.modal.direct/v1","name":"Modal","doc":"https://modal.com/docs/guide/endpoints","models":{"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM 5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.45,"output":1.5,"cache_read":0.09}},"thinkingmachines/Inkling-NVFP4":{"id":"thinkingmachines/Inkling-NVFP4","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.2,"output":5,"cache_read":0.27}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8-Max","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1010000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"reasoning":15,"cache_read":0.3}}}},"lucidquery":{"id":"lucidquery","env":["LUCIDQUERY_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lucidquery.com/v1","name":"LucidQuery","doc":"https://lucidquery.com/docs","models":{"lucidquery-nexus-coder":{"id":"lucidquery-nexus-coder","name":"LucidQuery Nexus Coder","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"lucid","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-08-01","release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":250000,"output":60000},"cost":{"input":2,"output":5}},"lucidquery-agi-01-frontier":{"id":"lucidquery-agi-01-frontier","name":"AGI-01 Frontier","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-06-05","release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":120000},"cost":{"input":4.5,"output":22}},"lucidquery-agi-01-swift":{"id":"lucidquery-agi-01-swift","name":"AGI-01 Swift","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-06-05","release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":120000},"cost":{"input":2.5,"output":15}},"lucidnova-rf1-100b":{"id":"lucidnova-rf1-100b","name":"LucidNova RF1 100B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2025-09-16","release_date":"2024-12-28","last_updated":"2025-09-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":120000,"output":8000},"cost":{"input":2,"output":5}}}},"atomic-chat":{"id":"atomic-chat","env":["ATOMIC_CHAT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:1337/v1","name":"Atomic Chat","doc":"https://atomic.chat","models":{"Meta-Llama-3_1-8B-Instruct-GGUF":{"id":"Meta-Llama-3_1-8B-Instruct-GGUF","name":"Meta Llama 3.1 8B Instruct (GGUF)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0,"output":0}},"Qwen3_5-9B-Q4_K_M":{"id":"Qwen3_5-9B-Q4_K_M","name":"Qwen 3.5 9B (Q4_K_M)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"Qwen3_5-9B-MLX-4bit":{"id":"Qwen3_5-9B-MLX-4bit","name":"Qwen 3.5 9B (MLX 4-bit)","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-05","last_updated":"2026-04-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma-4-E4B-it-MLX-4bit":{"id":"gemma-4-E4B-it-MLX-4bit","name":"Gemma 4 E4B Instruct (MLX 4-bit)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"gemma-4-E4B-it-IQ4_XS":{"id":"gemma-4-E4B-it-IQ4_XS","name":"Gemma 4 E4B Instruct (IQ4_XS)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}}}},"umans-ai":{"id":"umans-ai","env":["UMANS_AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.code.umans.ai/v1","name":"Umans AI","doc":"https://app.umans.ai/offers/code/docs/orgs","models":{"umans-glm-5.2":{"id":"umans-glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":405504,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"umans-kimi-k3":{"id":"umans-kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"umans-kimi-k2.7":{"id":"umans-kimi-k2.7","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"umans-deepseek-v4-flash-0731":{"id":"umans-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"umans-coder":{"id":"umans-coder","name":"Umans Coder","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"umans-flash":{"id":"umans-flash","name":"Umans Flash","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.15,"output":1,"cache_read":0.05}},"umans-deepseek-v4-pro-0813":{"id":"umans-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393215},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}}}},"sakana":{"id":"sakana","env":["SAKANA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.sakana.ai/v1","name":"Sakana AI","doc":"https://console.sakana.ai/models","models":{"fugu":{"id":"fugu","name":"Fugu","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"}},"fugu-ultra":{"id":"fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"fugu-ultra-20260615":{"id":"fugu-ultra-20260615","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"provider":{"shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana-namazu":{"id":"sakana-namazu","name":"Sakana Namazu","description":"Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}}}},"deepinfra":{"id":"deepinfra","env":["DEEPINFRA_API_KEY"],"npm":"@ai-sdk/deepinfra","name":"Deep Infra","doc":"https://deepinfra.com/models","models":{"ByteDance/Seed-2.0-mini":{"id":"ByteDance/Seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02,"tiers":[{"input":0.2,"output":0.8,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"ByteDance/Seed-2.0-code":{"id":"ByteDance/Seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":1,"output":6,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"ByteDance/Seed-2.0-pro":{"id":"ByteDance/Seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.5,"output":3,"cache_read":0.1,"tiers":[{"input":1,"output":6,"cache_read":0.2,"tier":{"type":"context","size":128000}}]}},"stepfun-ai/Step-3.7-Flash":{"id":"stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.32,"output":0.89}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.09,"output":0.18,"cache_read":0.018}},"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.24,"output":0.9,"cache_read":0.135}},"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp":{"id":"deepseek-ai/DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.015}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek-V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek-R1-0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":2.15,"cache_read":0.35}},"deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":64000},"cost":{"input":0.26,"output":0.38,"cache_read":0.13}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning":{"id":"nvidia/Nemotron-3-Nano-Omni-30B-A3B-Reasoning","name":"Nemotron 3 Nano Omni 30B A3B Reasoning","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.2,"output":0.8}},"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5":{"id":"nvidia/Llama-3.3-Nemotron-Super-49B-v1.5","name":"Llama 3.3 Nemotron Super 49B v1.5","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.4,"output":0.4}},"nvidia/Nemotron-3-Nano-30B-A3B":{"id":"nvidia/Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.13,"output":0.38}},"google/gemma-4-E4B-it":{"id":"google/gemma-4-E4B-it","name":"Gemma 4 E4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.02,"output":0.1}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.07,"output":0.34}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":1.05,"output":3.5,"cache_read":0.205}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.12}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.75,"output":2.4,"cache_read":0.14}},"zai-org/GLM-4.7-Flash":{"id":"zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"status":"deprecated","cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"status":"deprecated","cost":{"input":0.6,"output":2.08,"cache_read":0.12}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.5,"output":2,"cache_read":0.1}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"thinkingmachines/Inkling-Small":{"id":"thinkingmachines/Inkling-Small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"tiers":[{"input":5,"output":15,"cache_read":1,"tier":{"type":"context","size":32000}},{"input":6.25,"output":18.5,"cache_read":1.25,"tier":{"type":"context","size":128000}}]}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3,"cache_read":0.04}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.6}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-Turbo","name":"Qwen3 Coder 480B A35B Instruct Turbo","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":66536},"cost":{"input":0.3,"output":1,"cache_read":0.1}},"Qwen/Qwen3.8-Max":{"id":"Qwen/Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":1.65,"output":4.951,"cache_read":0.206}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.15}},"Qwen/Qwen3-30B-A3B":{"id":"Qwen/Qwen3-30B-A3B","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.12,"output":0.5}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.4}},"Qwen/Qwen3.8-2.4T-A95B":{"id":"Qwen/Qwen3.8-2.4T-A95B","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.2}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.55}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.88,"cache_read":0.11}},"Qwen/Qwen3-Next-80B-A3B-Instruct":{"id":"Qwen/Qwen3-Next-80B-A3B-Instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":1.1}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen 3.5 397B A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-01","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.45,"output":3,"cache_read":0.22}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen 3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-01","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.14,"output":1,"cache_read":0.05}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.08,"output":0.28}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":3.2}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":81920},"cost":{"input":0.1,"output":0.95}},"Qwen/Qwen3-Max":{"id":"Qwen/Qwen3-Max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"tiers":[{"input":2.4,"output":12,"cache_read":0.48,"tier":{"type":"context","size":32000}},{"input":3,"output":15,"cache_read":0.6,"tier":{"type":"context","size":128000}}]}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","cost":{"input":0.15,"output":1.15,"cache_read":0.03}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.28,"output":1.1,"cache_read":0.056}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"status":"deprecated","cost":{"input":0.25,"output":1,"cache_read":0.05}},"meta-llama/Llama-4-Scout-17B-16E-Instruct":{"id":"meta-llama/Llama-4-Scout-17B-16E-Instruct","name":"Llama 4 Scout 17B","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/Llama-3.3-70B-Instruct-Turbo":{"id":"meta-llama/Llama-3.3-70B-Instruct-Turbo","name":"Llama 3.3 70B Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":0.2,"output":0.8}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.03,"output":0.14}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.037,"output":0.17}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.68,"output":3.4,"cache_read":0.136}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.75,"output":3.5,"cache_read":0.15}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.85,"output":14.25,"cache_read":0.285}},"tencent/Hy3":{"id":"tencent/Hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"XiaomiMiMo/MiMo-V2.5-Pro":{"id":"XiaomiMiMo/MiMo-V2.5-Pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":16384},"cost":{"input":1,"output":3,"cache_read":0.2}},"XiaomiMiMo/MiMo-V2.5":{"id":"XiaomiMiMo/MiMo-V2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.4,"output":2,"cache_read":0.08}}}},"wafer.ai":{"id":"wafer.ai","env":["WAFER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://pass.wafer.ai/v1","name":"Wafer","doc":"https://docs.wafer.ai/wafer-pass","models":{"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"General Language Model 5.1 — high-quality bilingual (EN/ZH) generation with strong coding and reasoning capabilities.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.1,"cache_write":0}},"glm5.2-fast":{"id":"glm5.2-fast","name":"GLM5.2-Fast","description":"The same model served for high TPS.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":10.25,"cache_read":0.5,"cache_write":0}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.2,"output":4.1,"cache_read":0.2,"cache_write":0}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Kimi K2.6 sparse MoE model with a 262K context window. Available serverless and not included in standard Wafer Pass. Non-ZDR only: requests with `Wafer-ZDR: required` are rejected.","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.14,"output":4.8,"cache_read":0.19,"cache_write":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.33,"output":1.32,"cache_read":0.07,"cache_write":0,"tiers":[{"input":0.66,"output":2.64,"cache_read":0.13,"cache_write":0,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.66,"output":2.64,"cache_read":0.13,"cache_write":0}}}}},"kilo":{"id":"kilo","env":["KILO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.kilo.ai/api/gateway","name":"Kilo Gateway","doc":"https://kilo.ai","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.25,"output":3.75,"cache_read":0.125,"cache_write":1.5625}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.65,"output":3.25,"cache_read":0.13,"cache_write":0.8125}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen: Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.23,"output":2.3}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.1,"output":0.15}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.0975,"output":0.78}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.195,"output":0.975,"cache_read":0.039,"cache_write":0.24375}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen: Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.2275,"output":0.91}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_write":0.40625}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.195,"output":1.56}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen3.8 27B is an open-weight dense vision-language model from Qwen. It is suited for coding, professional workflows, research, multimodal interaction, and long-running agent tasks, with flexible thinking that can be...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.425,"output":2.55,"cache_read":0.085,"cache_write":0.53125}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.1625,"output":1.3}},"qwen/qwen3.5-plus-20260420":{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen: Qwen3.5 Plus 2026-04-20","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8,"cache_write":0.375}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.08,"output":0.28}},"qwen/qwen3.5-plus-02-15":{"id":"qwen/qwen3.5-plus-02-15","name":"Qwen: Qwen3.5 Plus 2026-02-15","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.26,"output":1.56}},"qwen/qwen-plus-2025-07-28":{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen: Qwen Plus 0728","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen: Qwen3 Coder 480B A35B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.975,"output":4.875}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen: Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-02-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":0.8,"output":1,"cache_read":0.4}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.2925,"output":1.4625}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen: Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.1495,"output":0.598}},"qwen/qwen3.5-flash-02-23":{"id":"qwen/qwen3.5-flash-02-23","name":"Qwen: Qwen3.5-Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.065,"output":0.26}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.39,"output":2.34}},"qwen/qwen3-vl-8b-thinking":{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen: Qwen3 VL 8B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.18,"output":2.1}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.7}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Qwen3.7 Flash is a vision-language reasoning model from Alibaba. It is suited for multimodal agents, visual coding, search, and computer interaction, with strengths in object recognition, spatial understanding, and real-world...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038}},"qwen/qwen3-30b-a3b-thinking-2507":{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen: Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.1,"output":0.9,"cache_read":0.05}},"qwen/qwen3-max-thinking":{"id":"qwen/qwen3-max-thinking","name":"Qwen: Qwen3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen3.8 Max 0902 is an updated snapshot of Qwen3.8 Max from Alibaba's Qwen team. It is a 2.4-trillion-parameter mixture-of-experts model that accepts text, image, and video input and returns text,...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"cache_read":0.156,"cache_write":0.975}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen: Qwen3 VL 8B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen: Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":0.13,"output":0.52}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"cache_read":0.052,"cache_write":0.325}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.08}},"qwen/qwen-2.5-coder-32b-instruct":{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.66,"output":1}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen: Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":0.13,"output":0.52}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":16384},"cost":{"input":0.13,"output":0.52}},"qwen/qwen-2.5-7b-instruct":{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen: Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.1,"output":0.2}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen3.8 Flash is a multimodal reasoning model from Alibaba. It is suited for coding assistance, agentic workflows, visual understanding, document and codebase analysis, desktop interaction, chart analysis, and long-video analysis.","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen: Qwen3 VL 30B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.027,"output":6.162,"cache_write":1.28375}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.36,"output":0.4}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen: Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.455,"output":1.82}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.26,"output":1.04}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Qwen3.7-Plus is a cost-effective model in Alibaba's Qwen3.7 series. It supports text and image input with text output, building on the series' text capabilities with a comprehensive upgrade to its...","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.032,"cache_write":0.4}},"qwen/qwen3-vl-32b-instruct":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen: Qwen3 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-23","last_updated":"2025-10-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.104,"output":0.416}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"Baidu: ERNIE 4.5 VL 424B A47B ","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"AionLabs: Aion-2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"AionLabs: Aion-RP 1.0 (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-02-04","last_updated":"2025-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.8,"output":1.6}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"AionLabs: Aion-3.0","description":"Aion-3.0 is a multi-model roleplaying and storytelling system from AionLabs, built on the GLM family of models. It uses a collaborative generation process in which multiple specialized models each contribute...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"AionLabs: Aion-3.0-Mini","description":"Aion-3.0 Mini is a multi-model roleplaying and storytelling system from AionLabs, built on the DeepSeek family of models. It uses a collaborative generation process in which multiple specialized models each...","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"~anthropic/claude-fable-latest":{"id":"~anthropic/claude-fable-latest","name":"Anthropic: Claude Fable Latest ($$$$)","description":"This model always redirects to the latest model in the Claude Fable family.","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"~anthropic/claude-opus-latest":{"id":"~anthropic/claude-opus-latest","name":"Anthropic: Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"~anthropic/claude-haiku-latest":{"id":"~anthropic/claude-haiku-latest","name":"Anthropic Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"~anthropic/claude-sonnet-latest":{"id":"~anthropic/claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph: Morph V3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.9,"output":1.9}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph: Morph V3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":38000},"cost":{"input":0.8,"output":1.2}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-07-22","last_updated":"2023-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":6144,"output":5529},"cost":{"input":0.35,"output":0.65}},"~deepseek/deepseek-v4-flash-latest":{"id":"~deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"This model always redirects to the latest model in the DeepSeek V4 Flash family.","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":393216},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"dots-studio/dots-3-note-preview:free":{"id":"dots-studio/dots-3-note-preview:free","name":"Dots Studio: Dots3-Note Preview (free)","description":"Dots3-Note Preview is an open-weight mixture-of-experts model from Dots Studio, with 16B active parameters out of 280B total. It is the lightest model in the Dots 3 family and is...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":460800},"cost":{"input":0,"output":0}},"~x-ai/grok-latest":{"id":"~x-ai/grok-latest","name":"xAI: Grok Latest","description":"This model always redirects to the latest Grok model from xAI.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"meituan/longcat-2.0":{"id":"meituan/longcat-2.0","name":"Meituan: LongCat 2.0","description":"LongCat 2.0 is a sparse mixture-of-experts language model from Meituan, with 48B active parameters out of 1.6T total. It is suited for coding, repository-level changes, long-horizon problem solving, and agentic...","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048756,"output":262144},"cost":{"input":0.75,"output":3,"cache_read":0.015}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Poolside: Laguna XS 2.1","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.2,"cache_read":0.05}},"poolside/laguna-xs-2.1:free":{"id":"poolside/laguna-xs-2.1:free","name":"Poolside: Laguna XS 2.1 (free)","description":"Laguna XS 2.1 is the latest coding agent model in the 33B-A3B category from [Poolside](https://poolside.ai/) and a step forward from their Laguna XS.2 model (released in April 2026). It combines...","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1:free":{"id":"poolside/laguna-s-2.1:free","name":"Poolside: Laguna S 2.1 (free)","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Poolside: Laguna S 2.1","description":"Laguna S 2.1 is the latest coding agent model from [Poolside](<https://poolside.ai/>). Laguna S 2.1 is a 118B total parameter model with 8B active parameters, scoring 70.2% on Terminal-Bench 2.1 and...","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.1,"output":0.2,"cache_read":0.01}},"kwaipilot/kat-coder-pro-v2":{"id":"kwaipilot/kat-coder-pro-v2","name":"Kwaipilot: KAT-Coder-Pro V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":144000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"Kwaipilot: KAT-Coder-Pro V2.5","description":"KAT-Coder-Pro V2.5 is a flagship-level Agentic Coding model that can directly hand over an entire issue or an entire business workflow to it, allowing it to autonomously locate and make...","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters...","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":230400},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash:free":{"id":"stepfun/step-3.7-flash:free","name":"StepFun: Step 3.7 Flash (free)","description":"Step 3.7 Flash is StepFun's latest high-efficiency multimodal Mixture-of-Experts model. It pairs a 196B-parameter language backbone with a vision encoder for native image and video understanding, activating roughly 11B parameters per token. The model supports a 256K context window and exposes selectable reasoning levels (high/medium/low), letting callers trade off speed, cost, and depth of reasoning. Designed for coding, agentic workflows, structured outputs, and long-context productivity tasks.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"reasoning":0,"cache_read":0}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Mistral: Ministral 3 14B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-02-26","last_updated":"2024-02-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Mistral: Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":204800},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"mistralai/mistral-medium-3-5":{"id":"mistralai/mistral-medium-3-5","name":"Mistral: Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":1.5,"output":7.5}},"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Devstral 2 is a state-of-the-art open-source model by Mistral AI specializing in agentic coding. It is a 123B-parameter dense transformer model supporting a 256K context window. Devstral 2 supports exploring...","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-large-2407":{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-19","last_updated":"2024-11-19","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral: Mistral Small 3.2 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.075,"output":0.2}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mistral: Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Mistral: Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.2,"output":0.6,"cache_read":0.02}},"mistralai/mistral-large-2512":{"id":"mistralai/mistral-large-2512","name":"Mistral Large 3","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Mistral: Ministral 3 3B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.1,"output":0.1,"cache_read":0.01}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.019,"output":0.03}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral: Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral: Mistral Small 3","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":16384},"cost":{"input":0.05,"output":0.08}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral: Mistral Small 3.1 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":0.351,"output":0.555}},"mistralai/mistral-small-2603":{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Mistral: Ministral 3 8B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.15,"cache_read":0.015}},"mistralai/voxtral-small-24b-2507":{"id":"mistralai/voxtral-small-24b-2507","name":"Mistral: Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral: Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.003,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.004,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax-M2 Her","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m1":{"id":"minimax/minimax-m1","name":"MiniMax: MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":40000},"cost":{"input":0.4,"output":2.2}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax: MiniMax-01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000192,"output":900172},"cost":{"input":0.2,"output":1.1}},"nvidia/nemotron-3.5-lightning:free":{"id":"nvidia/nemotron-3.5-lightning:free","name":"NVIDIA: Nemotron 3.5 Lightning (free)","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting...","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.2,"output":0.2}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"NVIDIA Nemotron 3.5 Lightning is an open mixture-of-experts model from NVIDIA, with 3B active parameters out of 30B total. It is suited for high-throughput agentic workloads and specialized tasks that...","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.08,"output":0.2,"cache_read":0.04}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"NVIDIA: Nemotron 3 Nano Omni (free)","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.085,"output":0.4}},"nvidia/nemotron-3-ultra-550b-a55b:free":{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"NVIDIA: Nemotron 3 Ultra (free)","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b:free":{"id":"nvidia/nemotron-3-super-120b-a12b:free","name":"NVIDIA: Nemotron 3 Super (free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety:free":{"id":"nvidia/nemotron-3.5-content-safety:free","name":"NVIDIA: Nemotron 3.5 Content Safety (free)","description":"NVIDIA Nemotron 3.5 Content Safety is a compact 4B-parameter multimodal guardrail model from NVIDIA, fine-tuned from Google Gemma-3-4B. It moderates both inputs to and responses from LLMs and VLMs, accepting... **Terms of service** For NVIDIA free endpoints (Super/Ultra/etc): Trial use only - do not submit personal or confidential data. Your use is logged for security purposes and to improve NVIDIA products and services. The logged session data for improvement purposes is not linked to your identity or any persistent identifier. For more information about our data processing practices, see our [Privacy Policy](https://www.nvidia.com/en-us/about-nvidia/privacy-policy/). By interacting with this endpoint, you consent to our collection, recording, and use of such information and the [NVIDIA API Trial Terms of Service](https://assets.ngc.nvidia.com/products/api-catalog/legal/NVIDIA%20API%20Trial%20Terms%20of%20Service.pdf).","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"NVIDIA Nemotron 3 Ultra is an open frontier-reasoning and orchestration model from NVIDIA, with 55B active parameters out of 550B total (MoE). Built on a hybrid Transformer-Mamba mixture-of-experts architecture, it...","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32768},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Claude Opus 4.8 is Anthropic's most capable generally available model in the Opus family. It supports text, image, and file inputs with text output, with reasoning support and a 1M-token...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Claude Opus 5 is Anthropic’s flagship model for demanding reasoning, coding, and long-horizon agentic work. It is particularly strong at end-to-end software tasks, code review and bug finding, visual analysis...","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Anthropic: Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude Fable 5 is a Mythos-class model from Anthropic, built for autonomous knowledge work and coding. It supports text, image, and file inputs with text output, with reasoning support and...","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Anthropic: Claude Opus 4 ($$$$)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Anthropic: Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Sonnet 5 is Anthropic's most capable Sonnet-class model, with frontier performance across coding, agents, and professional work. It supports adaptive thinking with selectable reasoning effort levels (low, medium, high, max,...","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude Fable 5.1 improves on Claude Fable 5 across the board, with the biggest gains in agentic coding, long-running agentic workflows, and knowledge work: long code refactors, front-end and visual...","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.042,"output":0.22}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Nano Banana 2 Lite (Gemini 3.1 Flash Lite Image) is Google's fastest, most cost-efficient Gemini image model, built for high-velocity developer pipelines and rapid-fire visual exploration. It delivers text-to-image generation...","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.25,"output":1.5}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Google: Gemma 3 4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.1}},"google/lyria-3-clip-preview":{"id":"google/lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.25,"cache_read":0.015,"cache_write":0.041667}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro is Google’s most advanced image-generation and editing model, built on Gemini 3 Pro. It extends the original Nano Banana with significantly improved multimodal reasoning, real-world grounding, and...","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":6,"reasoning":6,"cache_read":0.1,"cache_write":0.1875}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"reasoning":0.4,"cache_read":0.01,"cache_write":0.083333}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Gemini 3.6 Flash is a high-efficiency model from Google for coding, agentic workflows, and web and app development. It is designed to produce polished outputs with fewer unnecessary edits and...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.375,"output":1.875,"reasoning":1.875,"cache_read":0.0375,"cache_write":0.020833}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.125,"output":0.75,"reasoning":0.75,"cache_read":0.0125,"cache_write":0.041667}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":4.5,"reasoning":4.5,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.125,"output":0.75,"reasoning":0.75,"cache_read":0.0125,"cache_write":0.041667}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Google: Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.08,"output":0.16}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Gemini 3.1 Flash Image, a.k.a. \"Nano Banana 2,\" is Google’s latest state of the art image generation and editing model, delivering Pro-level visual quality at Flash speed. It combines advanced...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Gemini 3.5 Flash Lite is a high-efficiency model from Google with upgraded agentic capabilities. It is suited for subagents that execute focused tasks within complex, multi-agent workflows.","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.15,"output":1.25,"reasoning":1.25,"cache_read":0.015,"cache_write":0.041667}},"google/gemini-2.5-pro-preview":{"id":"google/gemini-2.5-pro-preview","name":"Google: Gemini 2.5 Pro Preview 06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["image","text","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":1,"output":6,"reasoning":6,"cache_read":0.1,"cache_write":0.1875}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.34,"cache_read":0.05}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.041667}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Gemini 3.8 Flash is Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows.","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/lyria-3-pro-preview":{"id":"google/lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Gemini 3.7 Flash is a multimodal model from Google for fast agentic workflows, coding, and complex multi-step reasoning. It is designed for tasks that require responsive performance and reliable multi-step...","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-2.5-pro-preview-05-06":{"id":"google/gemini-2.5-pro-preview-05-06","name":"Google: Gemini 2.5 Pro Preview 05-06","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Google: Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.15}},"google/gemma-2-27b-it":{"id":"google/gemma-2-27b-it","name":"Google: Gemma 2 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-07-13","last_updated":"2024-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":2048},"cost":{"input":0.65,"output":0.65}},"relace/relace-apply-3":{"id":"relace/relace-apply-3","name":"Relace: Relace Apply 3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.85,"output":1.25}},"relace/relace-search":{"id":"relace/relace-search","name":"Relace: Relace Search","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":1,"output":3}},"nex-agi/nex-n2.5-mini:free":{"id":"nex-agi/nex-n2.5-mini:free","name":"Nex AGI: Nex-N2.5-Mini (free)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nex-agi/nex-n2.5-pro:free":{"id":"nex-agi/nex-n2.5-pro:free","name":"Nex AGI: Nex-N2.5-Pro (free)","description":"Nex-N2.5 is an agentic model built to turn goals into working, verified outcomes. Its core strength is agentic coding within a visual feedback loop: it can explore codebases, implement multi-file...","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling-small:free":{"id":"thinkingmachines/inkling-small:free","name":"Thinking Machines: Inkling Small (free)","description":"Inkling Small is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 12B active parameters out of 276B total. It is positioned as the smaller, more efficient member of...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling:free":{"id":"thinkingmachines/inkling:free","name":"Thinking Machines: Inkling (free)","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Inkling is an open-weight multimodal mixture-of-experts model from Thinking Machines Lab, with 41B active parameters out of 975B total. It is designed for general-purpose reasoning, coding, agentic and tool-use systems,...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-07-02","last_updated":"2023-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":3686},"cost":{"input":0.06,"output":0.06}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It is designed to keep track of information across extended tasks, work through...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a reasoning model from Meta, designed for complex agentic tasks. It accepts text, images, video, audio, and PDF documents, returns text, and offers a 1M-token context...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Meta: Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 contributor tier is a reasoning model from Meta designed for developers who want to start building at an even lower cost. It’s meaningfully cheaper than Muse Spark...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Meta: Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 Contributor is the cost-efficient contributor tier of Meta’s multimodal reasoning model for experimentation, learning, and early-stage agentic, multi-agent, and coding workflows. It is designed to track information...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark 1.1 is a multimodal reasoning model from Meta, built for agentic tasks. It accepts text, images, video, audio, and PDF documents and returns text, with a 1M-token context...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer 30B is a dense, open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark and optimized for autonomous agents on consumer hardware. It is suited for long-horizon...","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":1.1,"cache_read":0.04}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron: Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.5}},"thedrummer/skyfall-36b-v2":{"id":"thedrummer/skyfall-36b-v2","name":"TheDrummer: Skyfall 36B V2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"thedrummer/unslopnemo-12b":{"id":"thedrummer/unslopnemo-12b","name":"TheDrummer: UnslopNemo 12B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-11-08","last_updated":"2024-11-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":819200},"cost":{"input":0.4,"output":0.4}},"thedrummer/cydonia-24b-v4.1":{"id":"thedrummer/cydonia-24b-v4.1","name":"TheDrummer: Cydonia 24B V4.1","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-09-27","last_updated":"2025-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"bytedance/ui-tars-1.5-7b":{"id":"bytedance/ui-tars-1.5-7b","name":"ByteDance: UI-TARS 7B ","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":2048},"cost":{"input":0.1,"output":0.2,"cache_read":0.1}},"bytedance-seed/seed-1.6-flash":{"id":"bytedance-seed/seed-1.6-flash","name":"ByteDance Seed: Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.3}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"ByteDance Seed: Seed 2.1 Turbo","description":"Seed 2.1 Turbo is a multimodal model from ByteDance Seed for coding and long-horizon agent workflows. It is suited for end-to-end software delivery, multi-step task execution, and understanding visual and...","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.5,"output":2.5}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"Seed 2.0 Code","description":"Seed 2.0 Code is a model from ByteDance Seed optimized for agentic coding. It is suited for frontend development, multilingual programming tasks, and coding-agent workflows in tools such as Claude...","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":3}},"bytedance-seed/seed-1.6":{"id":"bytedance-seed/seed-1.6","name":"ByteDance Seed: Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2}},"bytedance-seed/seed-2.0-mini":{"id":"bytedance-seed/seed-2.0-mini","name":"Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.1,"output":0.4}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.25,"output":2}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Inception: Mercury 2.5","description":"Mercury 2.5 is the fastest reasoning LLM, and the latest diffusion LLM (dLLM) from Inception. Instead of generating tokens sequentially, Mercury 2.5 produces and refines multiple tokens in parallel, achieving...","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.2,"output":0.75,"cache_read":0.02}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Inception: Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Writer: Palmyra X5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"palmyra","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-21","last_updated":"2026-01-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"~google/gemini-pro-latest":{"id":"~google/gemini-pro-latest","name":"Google Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["audio","pdf","image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"~google/gemini-flash-latest":{"id":"~google/gemini-flash-latest","name":"Google Gemini Flash Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"microsoft/phi-4":{"id":"microsoft/phi-4","name":"Microsoft: Phi 4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":14745},"cost":{"input":0.07,"output":0.14}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-04-16","last_updated":"2024-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Fugu Ultra is the higher-performance model in Sakana AI's Fugu family. Rather than a single monolithic model, Fugu is a learned multi-agent orchestration system: a language model trained to route...","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Sakana Namazu is a Japanese-specialized reasoning model from Sakana AI, based on Kimi K2.6 with additional training for Japanese language and business contexts. It is suited for Japanese instruction following,...","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"~moonshotai/kimi-latest":{"id":"~moonshotai/kimi-latest","name":"MoonshotAI Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":2.4,"output":12,"cache_read":0.24}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"IBM: Granite 4.2 8B","description":"Granite 4.2 8B is a dense reasoning model from IBM. It is suited for mathematics, code generation, multilingual dialogue, and agentic workflows that need multi-step reasoning. It supports full, low-effort,...","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.06,"output":0.25,"cache_read":0.015}},"ibm-granite/granite-4.0-h-micro":{"id":"ibm-granite/granite-4.0-h-micro","name":"IBM: Granite 4.0 Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":117900},"cost":{"input":0.017,"output":0.112}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek: DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4 Flash Vision Exp is an experimental vision-enabled version of [DeepSeek V4 Flash 0731](https://openrouter.ai/deepseek/deepseek-v4-flash-0731) from DeepSeek, adding image understanding while matching the base model on text capabilities including agents,...","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro 0813 is a large-scale mixture-of-experts model from DeepSeek. This is the GA release of DeepSeek V4 Pro.","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"DeepSeek V4 Flash 0731 is a sparse mixture-of-experts model from DeepSeek, with 13B active parameters out of 284B total. This re-post-trained revision is suited for coding, reasoning, and agent workflows.","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":0.44,"output":1.32,"cache_read":0.028}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash is a sparse mixture-of-experts model from DeepSeek, and the cost-efficient tier of the V4.1 family. DeepSeek reports that it exceeds V4 Pro on performance, speed, and task...","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16000},"cost":{"input":0.2574,"output":1.0287}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"DeepSeek: R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.5,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"DeepSeek: R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":7372},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek: DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek: DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":1.6,"output":3.2,"cache_read":0.135}},"deepseek/deepseek-chat-v3-0324":{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek: DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":147456},"cost":{"input":0.29,"output":1.14,"cache_read":0.11}},"~openai/gpt-latest":{"id":"~openai/gpt-latest","name":"OpenAI GPT Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"~openai/gpt-mini-latest":{"id":"~openai/gpt-mini-latest","name":"OpenAI GPT Mini Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Amazon: Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5}},"amazon/nova-micro-v1":{"id":"amazon/nova-micro-v1","name":"Amazon: Nova Micro 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":5120},"cost":{"input":0.035,"output":0.14}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Amazon: Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.8,"output":3.2}},"amazon/nova-premier-v1":{"id":"amazon/nova-premier-v1","name":"Amazon: Nova Premier 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":2.5,"output":12.5,"cache_read":0.625}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Amazon: Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.06,"output":0.24}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"inclusionAI: Ling 3.0 Flash","description":"*Ling-3.0-flash* is a *124B-parameter Mixture-of-Experts (MoE) model*, with approximately *5.1B parameters activated per token*. The model is designed with *token efficiency and production-scale agentic inference* as key priorities, enabling developers...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"inclusionAI: Ling 3.0 Flash Fin","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin:free":{"id":"inclusionai/ling-3.0-flash-fin:free","name":"inclusionAI: Ling 3.0 Flash Fin (free)","description":"Ling 3.0 Flash Fin is a finance-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for real-world investment...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante:free":{"id":"inclusionai/ling-3.0-flash-sante:free","name":"inclusionAI: Ling 3.0 Flash Sante (free)","description":"Ling 3.0 Flash Sante is a health and medicine-focused mixture-of-experts model from InclusionAI, built on Ling 3.0 Flash with 5.1B active parameters out of 124B total. It is designed for...","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl:free":{"id":"inclusionai/ling-3.0-flash-vl:free","name":"inclusionAI: Ling 3.0 Flash VL (free)","description":"Ling 3.0 Flash VL builds on Ling 3.0 Flash (124B total / 5.5B active MoE from InclusionAI), further strengthening its language capabilities while adding native visual perception and advanced visual...","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":2.5,"output":5}},"mancer/weaver":{"id":"mancer/weaver","name":"Mancer: Weaver (alpha)","description":"An attempt to recreate Claude-style verbosity, but don't expect the same level of coherence or memory. Meant for use in roleplay/narrative situations.","family":"alpha","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2023-08-02","last_updated":"2023-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":6000},"cost":{"input":0.4,"output":0.75}},"openrouter/free":{"id":"openrouter/free","name":"OpenRouter Free Models Router","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32768},"cost":{"input":0,"output":0}},"openrouter/pareto-code":{"id":"openrouter/pareto-code","name":"Pareto Code Router","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-05-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65536},"cost":{"input":0,"output":0}},"openrouter/bodybuilder":{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"status":"beta","cost":{"input":0,"output":0}},"openrouter/auto":{"id":"openrouter/auto","name":"Auto Router","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-15","last_updated":"2026-03-15","modalities":{"input":["audio","image","pdf","text","video"],"output":["image","text"]},"open_weights":false,"limit":{"context":2000000,"output":32768},"cost":{"input":0,"output":0}},"sao10k/l3.3-euryale-70b":{"id":"sao10k/l3.3-euryale-70b","name":"Sao10K: Llama 3.3 Euryale 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.65,"output":0.75}},"sao10k/l3-lunaris-8b":{"id":"sao10k/l3-lunaris-8b","name":"Sao10K: Llama 3 8B Lunaris","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":7372},"cost":{"input":0.04,"output":0.05}},"sao10k/l3.1-euryale-70b":{"id":"sao10k/l3.1-euryale-70b","name":"Sao10K: Llama 3.1 Euryale 70B v2.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-08-28","last_updated":"2024-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.85,"output":0.85}},"kilo-auto/free":{"id":"kilo-auto/free","name":"Auto Free","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":10000},"cost":{"input":0,"output":0,"reasoning":0,"cache_read":0,"cache_write":0}},"kilo-auto/efficient":{"id":"kilo-auto/efficient","name":"Auto Efficient","description":"Routes each request to the cheapest model that gets the job done, based on continuously benchmarked accuracy and cost.","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"reasoning":0,"cache_read":0.0325,"cache_write":0.40625}},"kilo-auto/small":{"id":"kilo-auto/small","name":"Auto Small","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.4,"reasoning":0,"cache_read":0.005}},"kilo-auto/frontier":{"id":"kilo-auto/frontier","name":"Auto Frontier","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"reasoning":0,"cache_read":0.5,"cache_write":6.25}},"kilo-auto/balanced":{"id":"kilo-auto/balanced","name":"Auto Balanced","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"1970-01-01","last_updated":"1970-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"reasoning":0,"cache_read":0.0325,"cache_write":0.40625}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"SpaceXAI: Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"SpaceXAI: Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"Grok 4.5 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.3}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Grok coding model for agentic engineering, edits, and codebase workflows","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"Grok 4.6 is SpaceXAI's smartest model with frontier performance on coding, knowledge work, and STEM.","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.02,"output":0.04}},"meta-llama/llama-guard-4-12b":{"id":"meta-llama/llama-guard-4-12b","name":"Meta: Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":16384},"cost":{"input":0.18,"output":0.18}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Meta: Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.33}},"meta-llama/llama-3.2-1b-instruct":{"id":"meta-llama/llama-3.2-1b-instruct","name":"Meta: Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":60000,"output":54000},"cost":{"input":0.027,"output":0.201}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Meta: Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":0.2,"output":0.696}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Meta: Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":327680,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/llama-3.1-70b-instruct":{"id":"meta-llama/llama-3.1-70b-instruct","name":"Llama-3.1-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.4,"output":0.4}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"nousresearch/hermes-3-llama-3.1-70b":{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Nous: Hermes 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-18","last_updated":"2024-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":0.7}},"nousresearch/hermes-3-llama-3.1-405b":{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Nous: Hermes 3 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2024-08-16","last_updated":"2024-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":1}},"nousresearch/hermes-4-405b":{"id":"nousresearch/hermes-4-405b","name":"Nous: Hermes 4 405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"nousresearch","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":1,"output":3}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"OpenAI: o4 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-2024-07-18":{"id":"openai/gpt-4o-mini-2024-07-18","name":"OpenAI: GPT-4o-mini (2024-07-18)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"OpenAI: o3 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-02-12","last_updated":"2025-02-12","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"OpenAI: GPT-6 Astra Pro ($$$$)","description":"GPT-6 Astra Pro is the same underlying model as [GPT-6 Astra](https://openrouter.ai/openai/gpt-6-astra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-audio-mini":{"id":"openai/gpt-audio-mini","name":"OpenAI: GPT Audio Mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio","pdf"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.4}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"GPT-5.6 Sol is the flagship model in OpenAI's GPT-5.6 series. It is suited for complex reasoning, coding, and agentic workflows, and is particularly strong at command-line and multi-step coding tasks...","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-4-turbo-preview":{"id":"openai/gpt-4-turbo-preview","name":"OpenAI: GPT-4 Turbo Preview ($$$$)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's flagship model for demanding end-to-end work. It is suited for advanced analysis, software engineering, deep research, scientific work, and document creation, with particular strengths in long-horizon...","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-5.2-chat":{"id":"openai/gpt-5.2-chat","name":"OpenAI: GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT-5.6 Luna","description":"GPT-5.6 Luna Pro is the same underlying model as [GPT-5.6 Luna](https://openrouter.ai/openai/gpt-5.6-luna), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.02,"output":0.1}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5-image":{"id":"openai/gpt-5-image","name":"OpenAI: GPT-5 Image ($$$$)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":10,"output":10,"cache_read":1.25}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT-5.6 Sol","description":"GPT-5.6 Sol Pro is the same underlying model as [GPT-5.6 Sol](https://openrouter.ai/openai/gpt-5.6-sol), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"OpenAI: gpt-oss-safeguard-20b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4-image-2":{"id":"openai/gpt-5.4-image-2","name":"OpenAI: GPT-5.4 Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":8,"output":15,"cache_read":2}},"openai/gpt-3.5-turbo-0613":{"id":"openai/gpt-3.5-turbo-0613","name":"OpenAI: GPT-3.5 Turbo (older v0613)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1,"output":2}},"openai/gpt-audio":{"id":"openai/gpt-audio","name":"OpenAI: GPT Audio","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio","pdf"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"GPT-5.6 Luna is a fast, cost-efficient model in OpenAI's GPT-5.6 series. It is suited for high-volume, latency-sensitive tasks such as chat, classification, and lightweight agentic workflows, providing capable reasoning for...","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT-5.6 Terra","description":"GPT-5.6 Terra Pro is the same underlying model as [GPT-5.6 Terra](https://openrouter.ai/openai/gpt-5.6-terra), served with `reasoning.mode` set to `pro` for higher-quality responses on complex tasks. Learn more in OpenAI's docs: https://developers.openai.com/api/docs/guides/reasoning#reasoning-mode","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"OpenAI: GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-05-05","last_updated":"2026-05-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo-16k":{"id":"openai/gpt-3.5-turbo-16k","name":"OpenAI: GPT-3.5 Turbo 16k","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2023-08-28","last_updated":"2023-08-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":3,"output":4}},"openai/gpt-5-image-mini":{"id":"openai/gpt-5-image-mini","name":"OpenAI: GPT-5 Image Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["pdf","image","text"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":2,"cache_read":0.25}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.03,"output":0.17,"cache_read":0.03}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"OpenAI: GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2023-09-28","last_updated":"2023-09-28","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1.5,"output":2}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"GPT-5.6 Terra is a balanced model in OpenAI's GPT-5.6 series, positioned between the flagship Sol tier and the cost-efficient Luna tier. It is suited for everyday coding, reasoning, and agentic...","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":4096},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-sol-discounted":{"id":"openai/gpt-5.6-sol-discounted","name":"OpenAI: GPT-5.6 Sol (50% off)","description":"GPT-5.6 Sol served by OpenAI through Vercel AI Gateway at 50% lower cost than other available inference providers. This promotion runs through September 18, 2026.","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"reasoning":0,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"~z-ai/glm-flash-latest":{"id":"~z-ai/glm-flash-latest","name":"Z.ai: GLM Flash Latest","description":"This model always redirects to the latest model in the GLM Flash family.","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015}},"~z-ai/glm-latest":{"id":"~z-ai/glm-latest","name":"Z.ai: GLM Latest","description":"This model always redirects to the latest GLM model from Z.ai.","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1024000,"output":128000},"cost":{"input":0.7,"output":2.2,"cache_read":0.13}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"MoonshotAI: Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":100352},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.8,"output":3.4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"MoonshotAI: Kimi K2.7 Code is a coding-focused model in Moonshot AI's Kimi K2 family, built to complete end-to-end programming tasks reliably over long contexts. It uses a native multimodal mixture-of-experts...","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":100352},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi K3 is a 2.8T parameter open-weight multimodal reasoning model from Moonshot AI. It is suited for complex coding, knowledge work, and long-horizon agentic workflows, and is particularly strong at...","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"MoonshotAI: Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":100352},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"cohere/north-mini-code:free":{"id":"cohere/north-mini-code:free","name":"Cohere: North Mini Code (free)","description":"North Mini Code is Cohere's first agentic coding model and the debut of its North family. A sparse mixture-of-experts model with 30B total parameters and 3B active, it is optimized...","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a":{"id":"cohere/command-a","name":"Cohere: Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Upstage: Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Upstage: Solar Pro 4","description":"Solar Pro 4 is a large language model from Upstage. It is suited for agentic workflows, office productivity, document-intensive work, and coding.","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":80000},"cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Hy3 is a 295B-parameter Mixture-of-Experts model from Tencent (21B active, 192 experts with top-8 routing) built for reasoning, agentic workflows, and real-world production use. It supports a configurable reasoning effort:...","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.0825,"output":0.33,"cache_read":0.020625}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"Tencent: Hy4 preview is a mixture-of-experts model from Tencent, with 49B active parameters out of 770B total. It is designed for coding agents, complex tool-use workflows, and productivity tasks that...","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-30b-a3b":{"id":"tencent/hy-mt2-30b-a3b","name":"Tencent: Hy-MT2-30B-A3B","description":"Hy-MT2-30B-A3B is Tencent's flagship translation model in the Hy-MT2 family. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and...","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-7b":{"id":"tencent/hy-mt2-7b","name":"Tencent: Hy-MT2-7B","description":"Hy-MT2-7B is a 7B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided translation.","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-1.8b":{"id":"tencent/hy-mt2-1.8b","name":"Tencent: Hy-MT2-1.8B","description":"Hy-MT2-1.8B is a compact 1.8B-parameter translation model from Tencent. It supports 33 language pairs and five Chinese dialect and minority-language pairs, with workflows for structured, delimiter-based, contextual, glossary-based, and style-guided...","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":4096},"cost":{"input":0.044,"output":0.177}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.18,"output":0.6,"cache_read":0.06}},"tencent/hunyuan-a13b-instruct":{"id":"tencent/hunyuan-a13b-instruct","name":"Tencent: Hunyuan A13B Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.14,"output":0.57}},"liquid/lfm-2.5-2.6b:free":{"id":"liquid/lfm-2.5-2.6b:free","name":"LiquidAI: LFM2.5-2.6B (free)","description":"LFM2.5-2.6B is a compact reasoning model from Liquid AI. It is suited for agent workflows, data extraction, RAG, and long-context processing. Liquid advises against using it for agentic coding or...","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"GLM 5.2 is a large-scale reasoning model from Z.ai. It supports text input and output with a 1M-token context window, and is suited for long-horizon agent workflows, project-level software engineering,...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"GLM-5.3-Flash is a native multimodal model from Z.ai. It is suited for efficient coding and long-horizon agent tasks. Its hybrid sparse and linear attention architecture maintains accurate long-context behavior while...","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":198000,"output":128000},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"GLM-5.3 is a large-scale reasoning model from Z.ai, built for complex software engineering and long-horizon agent tasks. It supports text input and output with a 1M-token context window, and improves...","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.0605,"output":0.4}},"cognitivecomputations/dolphin-mistral-24b-venice-edition":{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Venice: Uncensored","description":"Venice Uncensored Dolphin Mistral 24B Venice Edition is a fine-tuned variant of Mistral-Small-24B-Instruct-2501, developed by dphn.ai in collaboration with Venice.ai. This model is designed as an “uncensored” instruct-tuned LLM, preserving...","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"perplexity/sonar-pro-search":{"id":"perplexity/sonar-pro-search","name":"Perplexity: Sonar Pro Search","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Perplexity: Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":114364},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Perplexity: Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Perplexity: Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Perplexity: Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8,"reasoning":3}},"rekaai/reka-edge":{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"reka","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":14745},"cost":{"input":0.1,"output":0.1}},"rekaai/reka-flash-3":{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"reka","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.1,"output":0.2}},"stealth/claude-opus-4.8":{"id":"stealth/claude-opus-4.8","name":"Stealth: Claude Opus 4.8 (20% off)","description":"Your prompts and completions may be retained and used to train or improve the provider's services. This third-party-served variant of Claude Opus 4.8 is offered at 20% lower cost than standard Claude Opus 4.8 pricing and is not served by Anthropic or Kilo Code.","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/qwen3.6-plus":{"id":"stealth/qwen3.6-plus","name":"Stealth: Qwen3.6 Plus (50% off)","description":"Your prompts and completions may be retained and used to train or improve the provider's services. This third-party-served variant of Qwen3.6 Plus is offered at 50% lower cost than standard Qwen3.6 Plus pricing and is not served by Alibaba or Kilo Code. Note: a surcharge applies to long-context workloads exceeding 256K input tokens.","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":0,"cache_read":0.025,"cache_write":0.3125}},"stealth/claude-opus-4.7":{"id":"stealth/claude-opus-4.7","name":"Stealth: Claude Opus 4.7 (20% off)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}},"stealth/claude-sonnet-4.6":{"id":"stealth/claude-sonnet-4.6","name":"Stealth: Claude Sonnet 4.6 (20% off)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.4,"output":12,"reasoning":0,"cache_read":0.24,"cache_write":3}},"stealth/claude-opus-4.6":{"id":"stealth/claude-opus-4.6","name":"Stealth: Claude Opus 4.6 (20% off)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":4,"output":20,"reasoning":0,"cache_read":0.4,"cache_write":5}}}},"alibaba-coding-plan":{"id":"alibaba-coding-plan","env":["ALIBABA_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding-intl.dashscope.aliyuncs.com/v1","name":"Alibaba Coding Plan","doc":"https://www.alibabacloud.com/help/en/model-studio/coding-plan","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen3-coder-plus":{"id":"qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":24576},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3-max-2026-01-23":{"id":"qwen3-max-2026-01-23","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"submodel":{"id":"submodel","env":["SUBMODEL_INSTAGEN_ACCESS_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://llm.submodel.ai/v1","name":"submodel","doc":"https://submodel.gitbook.io","models":{"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.2,"output":0.8}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.2,"output":0.8}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":75000,"output":163840},"cost":{"input":0.5,"output":2.15}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.1,"output":0.5}},"zai-org/GLM-4.5-FP8":{"id":"zai-org/GLM-4.5-FP8","name":"GLM 4.5 FP8","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.2,"output":0.3}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":0.8}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.2,"output":0.6}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-23","last_updated":"2025-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.5}}}},"openreason":{"id":"openreason","env":["OPENREASON_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.openreason.app/v1","name":"OpenReason","doc":"https://openreason.app/docs","models":{"deepseek-ai/deepseek-v4-flash-0731":{"id":"deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.1371,"output":0.2743}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1055,"output":0.422}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.0022,"output":4.22}}}},"azure":{"id":"azure","env":["AZURE_RESOURCE_NAME","AZURE_API_KEY"],"npm":"@ai-sdk/azure","name":"Azure","doc":"https://learn.microsoft.com/en-us/azure/ai-services/openai/concepts/models","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"mistral-small-2503":{"id":"mistral-small-2503","name":"Mistral Small 3.1","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":0.1,"output":0.3}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"ministral-3b":{"id":"ministral-3b","name":"Ministral 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.04}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-3.5-turbo-1106":{"id":"gpt-3.5-turbo-1106","name":"GPT-3.5 Turbo 1106","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":1,"output":2}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":272000},"cost":{"input":15,"output":120}},"phi-4-mini-reasoning":{"id":"phi-4-mini-reasoning","name":"Phi-4-mini-reasoning","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"codex-mini":{"id":"codex-mini","name":"Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-04","release_date":"2025-05-16","last_updated":"2025-05-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.5,"output":6,"cache_read":0.375}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"phi-4-reasoning-plus":{"id":"phi-4-reasoning-plus","name":"Phi-4-reasoning-plus","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4}},"cohere-command-a":{"id":"cohere-command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":2.5,"output":10}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"llama-4-maverick-17b-128e-instruct-fp8":{"id":"llama-4-maverick-17b-128e-instruct-fp8","name":"Llama 4 Maverick 17B 128E Instruct FP8","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.25,"output":1}},"cohere-embed-v3-english":{"id":"cohere-embed-v3-english","name":"Embed v3 English","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek-V4-Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.19,"output":0.51}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"cohere-embed-v-4-0":{"id":"cohere-embed-v-4-0","name":"Embed v4","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":1536},"cost":{"input":0.12,"output":0}},"phi-4":{"id":"phi-4","name":"Phi-4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.125,"output":0.5}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-3.5-turbo-0125":{"id":"gpt-3.5-turbo-0125","name":"GPT-3.5 Turbo 0125","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16384,"output":16384},"status":"deprecated","cost":{"input":0.5,"output":1.5}},"phi-4-multimodal":{"id":"phi-4-multimodal","name":"Phi-4-multimodal","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"phi","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.08,"output":0.32,"input_audio":4}},"phi-4-mini":{"id":"phi-4-mini","name":"Phi-4-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"phi","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.075,"output":0.3}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"Grok 4.1 Fast (Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-27","last_updated":"2025-06-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":8192},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image","audio"],"output":["text","image","audio"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-07-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"status":"deprecated","cost":{"input":1.35,"output":5.4}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"gpt-4-turbo-vision":{"id":"gpt-4-turbo-vision","name":"GPT-4 Turbo Vision","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"llama-4-scout-17b-16e-instruct":{"id":"llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.78}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-image-1.5":{"id":"gpt-image-1.5","name":"GPT-Image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":32,"cache_read":1.25}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":2,"output":8,"cache_read":0.5}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-image-1":{"id":"gpt-image-1","name":"GPT-Image-1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":40,"cache_read":1.25}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"model-router":{"id":"model-router","name":"Model Router","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"model-router","attachment":true,"reasoning":false,"tool_call":true,"release_date":"2025-05-19","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384},"cost":{"input":0.14,"output":0}},"gpt-chat-latest":{"id":"gpt-chat-latest","name":"GPT Chat Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":111616,"output":16384},"status":"beta","cost":{"input":5,"output":30,"cache_read":0.5}},"cohere-embed-v3-multilingual":{"id":"cohere-embed-v3-multilingual","name":"Embed v3 Multilingual","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"cohere-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-07","last_updated":"2023-11-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512,"output":1024},"cost":{"input":0.1,"output":0}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"status":"beta"},"deepseek-v3.2-speciale":{"id":"deepseek-v3.2-speciale","name":"DeepSeek-V3.2-Speciale","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.58,"output":1.68}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-image-2":{"id":"gpt-image-2","name":"GPT-Image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":0.6,"output":3}},"phi-4-reasoning":{"id":"phi-4-reasoning","name":"Phi-4-reasoning","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"phi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2023-10","release_date":"2024-12-11","last_updated":"2024-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.125,"output":0.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2025-12-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek-V4-Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/models","shape":"completions"},"cost":{"input":1.74,"output":3.48}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"codestral-2501":{"id":"codestral-2501","name":"Codestral 25.01","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-03","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3,"output":0.9}},"gpt-3.5-turbo-instruct":{"id":"gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2021-08","release_date":"2023-09-21","last_updated":"2023-09-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4096,"output":4096},"status":"deprecated","cost":{"input":1.5,"output":2}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"grok-4-20-non-reasoning":{"id":"grok-4-20-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":8192},"status":"beta","cost":{"input":2,"output":6}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.125}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.71,"output":0.71}},"grok-4-20-reasoning":{"id":"grok-4-20-reasoning","name":"Grok 4.20 (Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":8192},"status":"beta","cost":{"input":2,"output":6}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"claude-mythos-5":{"id":"claude-mythos-5","name":"Claude Mythos 5","description":"Restricted Claude model for advanced cybersecurity and biology research workflows","family":"claude-mythos","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","provider":{"npm":"@ai-sdk/anthropic","api":"https://${AZURE_RESOURCE_NAME}.services.ai.azure.com/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}}}},"amazon-bedrock":{"id":"amazon-bedrock","env":["AWS_ACCESS_KEY_ID","AWS_SECRET_ACCESS_KEY","AWS_REGION","AWS_BEARER_TOKEN_BEDROCK"],"npm":"@ai-sdk/amazon-bedrock","name":"Amazon Bedrock","doc":"https://docs.aws.amazon.com/bedrock/latest/userguide/models-supported.html","models":{"moonshotai.kimi-k2.5":{"id":"moonshotai.kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262143,"output":16000},"cost":{"input":0.6,"output":3}},"global.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"global.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (Global)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"us.anthropic.claude-opus-5":{"id":"us.anthropic.claude-opus-5","name":"Claude Opus 5 (US)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.amazon.nova-pro-v1:0":{"id":"eu.amazon.nova-pro-v1:0","name":"Nova Pro (EU)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.92,"output":3.68,"cache_read":0.23}},"us.writer.palmyra-x4-v1:0":{"id":"us.writer.palmyra-x4-v1:0","name":"Palmyra X4 (US)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":122880,"output":8192},"cost":{"input":2.5,"output":10}},"us.anthropic.claude-opus-4-6-v1":{"id":"us.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (US)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"us.xai.grok-4.6":{"id":"us.xai.grok-4.6","name":"Grok 4.6 (US)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2.2,"output":6.6,"cache_read":0.55}},"eu.mistral.pixtral-large-2502-v1:0":{"id":"eu.mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (EU)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"qwen.qwen3-coder-next":{"id":"qwen.qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.22,"output":1.8}},"global.openai.gpt-5.6-luna":{"id":"global.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (Global)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"global.anthropic.claude-opus-4-6-v1":{"id":"global.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (Global)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai.gpt-5.5":{"id":"openai.gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":5.5,"output":33,"cache_read":0.55}},"qwen.qwen3-coder-30b-a3b-v1:0":{"id":"qwen.qwen3-coder-30b-a3b-v1:0","name":"Qwen3 Coder 30B A3B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.6}},"global.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"global.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (Global)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"qwen.qwen3-235b-a22b-2507-v1:0":{"id":"qwen.qwen3-235b-a22b-2507-v1:0","name":"Qwen3 235B A22B 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.22,"output":0.88}},"mistral.ministral-3-3b-instruct":{"id":"mistral.ministral-3-3b-instruct","name":"Ministral 3 3B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.1,"output":0.1}},"global.anthropic.claude-sonnet-4-6":{"id":"global.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (Global)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"openai.gpt-5.4":{"id":"openai.gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-06-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.75,"output":16.5,"cache_read":0.275}},"mistral.pixtral-large-2502-v1:0":{"id":"mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"mistral.mistral-large-3-675b-instruct":{"id":"mistral.mistral-large-3-675b-instruct","name":"Mistral Large 3","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.5,"output":1.5}},"anthropic.claude-opus-4-5-20251101-v1:0":{"id":"anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"us.amazon.nova-micro-v1:0":{"id":"us.amazon.nova-micro-v1:0","name":"Nova Micro (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"jp.anthropic.claude-opus-4-7":{"id":"jp.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (JP)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.anthropic.claude-sonnet-5":{"id":"eu.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (EU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.22,"cache_write":2.75}},"apac.amazon.nova-micro-v1:0":{"id":"apac.amazon.nova-micro-v1:0","name":"Nova Micro (APAC)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.037,"output":0.148,"cache_read":0.00925}},"nvidia.nemotron-nano-9b-v2":{"id":"nvidia.nemotron-nano-9b-v2","name":"NVIDIA Nemotron Nano 9B v2","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.06,"output":0.23}},"au.anthropic.claude-sonnet-4-6":{"id":"au.anthropic.claude-sonnet-4-6","name":"AU Anthropic Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"anthropic.claude-opus-4-7":{"id":"anthropic.claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistral.ministral-3-8b-instruct":{"id":"mistral.ministral-3-8b-instruct","name":"Ministral 3 8B","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.15,"output":0.15}},"au.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"au.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (AU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"us.openai.gpt-5.6-sol":{"id":"us.openai.gpt-5.6-sol","name":"GPT-5.6 Sol (US)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5,"tiers":[{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11}}},"eu.amazon.nova-lite-v1:0":{"id":"eu.amazon.nova-lite-v1:0","name":"Nova Lite (EU)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.069,"output":0.276,"cache_read":0.01725}},"anthropic.claude-opus-5":{"id":"anthropic.claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.anthropic.claude-opus-4-6-v1":{"id":"eu.anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6 (EU)","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"apac.amazon.nova-pro-v1:0":{"id":"apac.amazon.nova-pro-v1:0","name":"Nova Pro (APAC)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.84,"output":3.36,"cache_read":0.21}},"anthropic.claude-sonnet-4-6":{"id":"anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"apac.amazon.nova-lite-v1:0":{"id":"apac.amazon.nova-lite-v1:0","name":"Nova Lite (APAC)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.063,"output":0.252,"cache_read":0.01575}},"mistral.voxtral-mini-3b-2507":{"id":"mistral.voxtral-mini-3b-2507","name":"Voxtral Mini 3B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["audio","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.04,"output":0.04}},"nvidia.nemotron-nano-12b-v2":{"id":"nvidia.nemotron-nano-12b-v2","name":"NVIDIA Nemotron Nano 12B v2 VL BF16","description":"Nemotron multimodal model for visual reasoning and agentic AI workflows","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.6}},"nvidia.nemotron-nano-3-30b":{"id":"nvidia.nemotron-nano-3-30b","name":"NVIDIA Nemotron Nano 3 30B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.06,"output":0.24}},"eu.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"eu.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"minimax.minimax-m2.1":{"id":"minimax.minimax-m2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"meta.llama3-3-70b-instruct-v1:0":{"id":"meta.llama3-3-70b-instruct-v1:0","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"deepseek.v3-v1:0":{"id":"deepseek.v3-v1:0","name":"DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.58,"output":1.68}},"eu.anthropic.claude-opus-5":{"id":"eu.anthropic.claude-opus-5","name":"Claude Opus 5 (EU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"anthropic.claude-sonnet-5":{"id":"anthropic.claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"us.writer.palmyra-x5-v1:0":{"id":"us.writer.palmyra-x5-v1:0","name":"Palmyra X5 (US)","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"us.meta.llama4-maverick-17b-instruct-v1:0":{"id":"us.meta.llama4-maverick-17b-instruct-v1:0","name":"Llama 4 Maverick 17B Instruct (US)","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.24,"output":0.97}},"meta.llama3-1-8b-instruct-v1:0":{"id":"meta.llama3-1-8b-instruct-v1:0","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.22,"output":0.22}},"minimax.minimax-m2":{"id":"minimax.minimax-m2","name":"MiniMax M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204608,"output":128000},"cost":{"input":0.3,"output":1.2}},"global.anthropic.claude-opus-5":{"id":"global.anthropic.claude-opus-5","name":"Claude Opus 5 (Global)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"qwen.qwen3-32b-v1:0":{"id":"qwen.qwen3-32b-v1:0","name":"Qwen3 32B (dense)","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.15,"output":0.6}},"writer.palmyra-x4-v1:0":{"id":"writer.palmyra-x4-v1:0","name":"Palmyra X4","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":122880,"output":8192},"cost":{"input":2.5,"output":10}},"us.amazon.nova-pro-v1:0":{"id":"us.amazon.nova-pro-v1:0","name":"Nova Pro (US)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"google.gemma-3-12b-it":{"id":"google.gemma-3-12b-it","name":"Google Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.049999999999999996,"output":0.09999999999999999}},"au.anthropic.claude-opus-4-8":{"id":"au.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (AU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"jp.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"jp.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (JP)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"eu.amazon.nova-2-lite-v1:0":{"id":"eu.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (EU)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.374,"output":3.157,"cache_read":0.0935}},"eu.anthropic.claude-opus-4-8":{"id":"eu.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (EU)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"jp.anthropic.claude-opus-5":{"id":"jp.anthropic.claude-opus-5","name":"Claude Opus 5 (JP)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"mistral.ministral-3-14b-instruct":{"id":"mistral.ministral-3-14b-instruct","name":"Ministral 14B 3.0","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.2,"output":0.2}},"openai.gpt-oss-safeguard-20b":{"id":"openai.gpt-oss-safeguard-20b","name":"GPT OSS Safeguard 20B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.2}},"global.amazon.nova-2-lite-v1:0":{"id":"global.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (Global)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.3,"output":2.5,"cache_read":0.075}},"eu.amazon.nova-micro-v1:0":{"id":"eu.amazon.nova-micro-v1:0","name":"Nova Micro (EU)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.04,"output":0.16,"cache_read":0.01}},"openai.gpt-5.6-luna":{"id":"openai.gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"anthropic.claude-opus-4-6-v1":{"id":"anthropic.claude-opus-4-6-v1","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"openai.gpt-oss-20b-1:0":{"id":"openai.gpt-oss-20b-1:0","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.07,"output":0.3}},"qwen.qwen3-vl-235b-a22b":{"id":"qwen.qwen3-vl-235b-a22b","name":"Qwen/Qwen3-VL-235B-A22B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-04","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.3,"output":1.5}},"amazon.nova-2-lite-v1:0":{"id":"amazon.nova-2-lite-v1:0","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.33,"output":2.75}},"global.xai.grok-4.6":{"id":"global.xai.grok-4.6","name":"Grok 4.6 (Global)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"global.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"global.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (Global)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"amazon.nova-lite-v1:0":{"id":"amazon.nova-lite-v1:0","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"anthropic.claude-opus-4-8":{"id":"anthropic.claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"us.amazon.nova-2-lite-v1:0":{"id":"us.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (US)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.33,"output":2.75,"cache_read":0.0825}},"us.openai.gpt-5.6-terra":{"id":"us.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (US)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"us.meta.llama3-3-70b-instruct-v1:0":{"id":"us.meta.llama3-3-70b-instruct-v1:0","name":"Llama 3.3 70B Instruct (US)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"us.meta.llama3-1-70b-instruct-v1:0":{"id":"us.meta.llama3-1-70b-instruct-v1:0","name":"Llama 3.1 70B Instruct (US)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"amazon.nova-pro-v1:0":{"id":"amazon.nova-pro-v1:0","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.8,"output":3.2,"cache_read":0.2}},"us.anthropic.claude-opus-4-7":{"id":"us.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"au.anthropic.claude-opus-4-6-v1":{"id":"au.anthropic.claude-opus-4-6-v1","name":"AU Anthropic Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":16.5,"output":82.5,"cache_read":1.65,"cache_write":20.625}},"writer.palmyra-x5-v1:0":{"id":"writer.palmyra-x5-v1:0","name":"Palmyra X5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"palmyra","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"global.openai.gpt-5.6-sol":{"id":"global.openai.gpt-5.6-sol","name":"GPT-5.6 Sol (Global)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai.gpt-5.6-sol":{"id":"openai.gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":4.4,"output":22,"cache_read":0.44,"cache_write":5.5,"tiers":[{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8.8,"output":33,"cache_read":0.88,"cache_write":11}}},"global.anthropic.claude-opus-4-8":{"id":"global.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (Global)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax.minimax-m2.5":{"id":"minimax.minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":98304},"cost":{"input":0.3,"output":1.2}},"openai.gpt-oss-120b-1:0":{"id":"openai.gpt-oss-120b-1:0","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"eu.anthropic.claude-opus-4-7":{"id":"eu.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (EU)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"us.meta.llama4-scout-17b-instruct-v1:0":{"id":"us.meta.llama4-scout-17b-instruct-v1:0","name":"Llama 4 Scout 17B Instruct (US)","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":3500000,"output":16384},"cost":{"input":0.17,"output":0.66}},"us.openai.gpt-5.6-luna":{"id":"us.openai.gpt-5.6-luna","name":"GPT-5.6 Luna (US)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.22,"output":1.32,"cache_read":0.022,"cache_write":0.275,"tiers":[{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.44,"output":1.98,"cache_read":0.044,"cache_write":0.55}}},"moonshot.kimi-k2-thinking":{"id":"moonshot.kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262143,"output":16000},"cost":{"input":0.6,"output":2.5}},"anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"deepseek.r1-v1:0":{"id":"deepseek.r1-v1:0","name":"DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"mistral.magistral-small-2509":{"id":"mistral.magistral-small-2509","name":"Magistral Small 1.2","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":40000},"cost":{"input":0.5,"output":1.5}},"us.anthropic.claude-fable-5":{"id":"us.anthropic.claude-fable-5","name":"Claude Fable 5 (US)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"eu.anthropic.claude-fable-5":{"id":"eu.anthropic.claude-fable-5","name":"Claude Fable 5 (EU)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75}},"us.openai.gpt-6-astra":{"id":"us.openai.gpt-6-astra","name":"GPT-6 Astra (US)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75,"tiers":[{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5}}},"us.anthropic.claude-fable-5-1":{"id":"us.anthropic.claude-fable-5-1","name":"Claude Fable 5.1 (US)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":11,"output":55,"cache_read":0.275,"cache_write":13.75}},"meta.llama4-scout-17b-instruct-v1:0":{"id":"meta.llama4-scout-17b-instruct-v1:0","name":"Llama 4 Scout 17B Instruct","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":3500000,"output":16384},"cost":{"input":0.17,"output":0.66}},"jp.amazon.nova-2-lite-v1:0":{"id":"jp.amazon.nova-2-lite-v1:0","name":"Nova 2 Lite (JP)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.396,"output":3.311,"cache_read":0.099}},"google.gemma-3-27b-it":{"id":"google.gemma-3-27b-it","name":"Google Gemma 3 27B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-27","last_updated":"2025-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":8192},"cost":{"input":0.12,"output":0.2}},"amazon.nova-micro-v1:0":{"id":"amazon.nova-micro-v1:0","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.035,"output":0.14,"cache_read":0.00875}},"us.mistral.pixtral-large-2502-v1:0":{"id":"us.mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (US)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"anthropic.claude-fable-5":{"id":"anthropic.claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"xai.grok-4.6":{"id":"xai.grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":6.6,"cache_read":0.55}},"global.anthropic.claude-fable-5":{"id":"global.anthropic.claude-fable-5","name":"Claude Fable 5 (Global)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"au.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"au.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (AU)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"eu.anthropic.claude-sonnet-4-6":{"id":"eu.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (EU)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"jp.anthropic.claude-opus-4-8":{"id":"jp.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (JP)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"eu.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"eu.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (EU)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1.1,"output":5.5,"cache_read":0.11,"cache_write":1.375}},"qwen.qwen3-next-80b-a3b":{"id":"qwen.qwen3-next-80b-a3b","name":"Qwen/Qwen3-Next-80B-A3B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.14,"output":1.4}},"us.anthropic.claude-sonnet-5":{"id":"us.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (US)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"us.amazon.nova-lite-v1:0":{"id":"us.amazon.nova-lite-v1:0","name":"Nova Lite (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.06,"output":0.24,"cache_read":0.015}},"global.anthropic.claude-opus-4-7":{"id":"global.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (Global)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"qwen.qwen3-coder-480b-a35b-v1:0":{"id":"qwen.qwen3-coder-480b-a35b-v1:0","name":"Qwen3 Coder 480B A35B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-09-18","last_updated":"2025-09-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.22,"output":1.8}},"openai.gpt-5.6-terra":{"id":"openai.gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":2.2,"output":13.2,"cache_read":0.22,"cache_write":2.75,"tiers":[{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4.4,"output":19.8,"cache_read":0.44,"cache_write":5.5}}},"nvidia.nemotron-super-3-120b":{"id":"nvidia.nemotron-super-3-120b","name":"NVIDIA Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.65}},"zai.glm-4.7-flash":{"id":"zai.glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4}},"google.gemma-3-4b-it":{"id":"google.gemma-3-4b-it","name":"Gemma 3 4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-01","last_updated":"2024-12-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.04,"output":0.08}},"global.openai.gpt-5.6-terra":{"id":"global.openai.gpt-5.6-terra","name":"GPT-5.6 Terra (Global)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"zai.glm-5":{"id":"zai.glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":101376},"cost":{"input":1,"output":3.2}},"openai.gpt-oss-safeguard-120b":{"id":"openai.gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"mistral.devstral-2-123b":{"id":"mistral.devstral-2-123b","name":"Devstral 2 123B","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":0.4,"output":2}},"openai.gpt-6-astra":{"id":"openai.gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":11,"output":55,"cache_read":1.1,"cache_write":13.75,"tiers":[{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":22,"output":82.5,"cache_read":2.2,"cache_write":27.5}}},"us.anthropic.claude-opus-4-1-20250805-v1:0":{"id":"us.anthropic.claude-opus-4-1-20250805-v1:0","name":"Claude Opus 4.1 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"us.anthropic.claude-sonnet-4-6":{"id":"us.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (US)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"mistral.voxtral-small-24b-2507":{"id":"mistral.voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.15,"output":0.35}},"openai.gpt-oss-20b":{"id":"openai.gpt-oss-20b","name":"gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/v1","shape":"responses"},"cost":{"input":0.07,"output":0.3}},"meta.llama4-maverick-17b-instruct-v1:0":{"id":"meta.llama4-maverick-17b-instruct-v1:0","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":16384},"cost":{"input":0.24,"output":0.97}},"zai.glm-4.7":{"id":"zai.glm-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"ca.amazon.nova-lite-v1:0":{"id":"ca.amazon.nova-lite-v1:0","name":"Nova Lite (CA)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.064,"output":0.256,"cache_read":0.016}},"us.anthropic.claude-opus-4-5-20251101-v1:0":{"id":"us.anthropic.claude-opus-4-5-20251101-v1:0","name":"Claude Opus 4.5 (US)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-08-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"au.anthropic.claude-opus-4-7":{"id":"au.anthropic.claude-opus-4-7","name":"Claude Opus 4.7 (AU)","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.5,"cache_read":0.55,"cache_write":6.875}},"jp.anthropic.claude-sonnet-4-6":{"id":"jp.anthropic.claude-sonnet-4-6","name":"Claude Sonnet 4.6 (JP)","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"us.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"us.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (US)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"us.deepseek.r1-v1:0":{"id":"us.deepseek.r1-v1:0","name":"DeepSeek-R1 (US)","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":1.35,"output":5.4}},"us.anthropic.claude-opus-4-8":{"id":"us.anthropic.claude-opus-4-8","name":"Claude Opus 4.8 (US)","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"au.anthropic.claude-opus-5":{"id":"au.anthropic.claude-opus-5","name":"Claude Opus 5 (AU)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic.claude-opus-4-1-20250805-v1:0":{"id":"anthropic.claude-opus-4-1-20250805-v1:0","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"jp.anthropic.claude-sonnet-5":{"id":"jp.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (JP)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"au.anthropic.claude-sonnet-5":{"id":"au.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (AU)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"xai.grok-4.3":{"id":"xai.grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-06-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"openai.gpt-oss-120b":{"id":"openai.gpt-oss-120b","name":"gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/amazon-bedrock/mantle","api":"https://bedrock-mantle.${AWS_REGION}.api.aws/v1","shape":"responses"},"cost":{"input":0.15,"output":0.6}},"meta.llama3-1-70b-instruct-v1:0":{"id":"meta.llama3-1-70b-instruct-v1:0","name":"Llama 3.1 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.72,"output":0.72}},"global.anthropic.claude-fable-5-1":{"id":"global.anthropic.claude-fable-5-1","name":"Claude Fable 5.1 (Global)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"us.anthropic.claude-haiku-4-5-20251001-v1:0":{"id":"us.anthropic.claude-haiku-4-5-20251001-v1:0","name":"Claude Haiku 4.5 (US)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic.claude-fable-5-1":{"id":"anthropic.claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"global.anthropic.claude-sonnet-5":{"id":"global.anthropic.claude-sonnet-5","name":"Claude Sonnet 5 (Global)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"us.meta.llama3-1-8b-instruct-v1:0":{"id":"us.meta.llama3-1-8b-instruct-v1:0","name":"Llama 3.1 8B Instruct (US)","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.22,"output":0.22}},"jp.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"jp.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (JP)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"global.openai.gpt-6-astra":{"id":"global.openai.gpt-6-astra","name":"GPT-6 Astra (Global)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"eu.anthropic.claude-sonnet-4-5-20250929-v1:0":{"id":"eu.anthropic.claude-sonnet-4-5-20250929-v1:0","name":"Claude Sonnet 4.5 (EU)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3.3,"output":16.5,"cache_read":0.33,"cache_write":4.125}},"deepseek.v3.2":{"id":"deepseek.v3.2","name":"DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.62,"output":1.85}}}},"merge-gateway":{"id":"merge-gateway","env":["MERGE_GATEWAY_API_KEY"],"npm":"merge-gateway-ai-sdk-provider","api":"https://api-gateway.merge.dev/v1/ai-sdk","name":"Merge Gateway","doc":"https://docs.merge.dev/merge-gateway","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.825,"output":2.4755,"cache_read":0.165}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.574,"output":2.294,"cache_read":0.1148}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.09,"output":0.13}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.144,"output":0.574,"cache_read":0.0288}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.144,"output":0.574,"cache_read":0.0288}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.276,"output":1.651,"cache_read":0.0552}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.086,"output":0.688,"cache_read":0.0172}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.057,"output":0.459,"cache_read":0.020357}},"qwen/qwen-flash":{"id":"qwen/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.022,"output":0.216,"cache_read":0.0044}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.029,"output":0.287,"cache_read":0.0058}},"qwen/qwen3-vl-plus":{"id":"qwen/qwen3-vl-plus","name":"Qwen3-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.143,"output":1.434,"cache_read":0.0286}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.15,"output":0.8,"cache_read":0.075}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":1.032,"cache_read":0.0344}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.289,"output":2.4}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.248,"output":1.485,"cache_read":0.0496}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.359,"output":1.434,"cache_read":0.0718}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":1010000},"cost":{"input":2.5,"output":6.25,"cache_read":0.5}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.115,"output":0.287,"cache_read":0.023}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.115,"output":0.917,"cache_read":0.023}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.165,"output":0.99,"cache_read":0.033}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.108,"output":1.076,"cache_read":0.0216}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"cost":{"input":1.31,"output":7.88}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":1.147,"cache_read":0.0574}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3-VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":2.867,"cache_read":0.0574}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3-VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.287,"output":1.147,"cache_read":0.15785}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.6,"cache_read":0.05}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3 Coder 480B A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.22,"output":1.8}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.115,"output":0.688,"cache_read":0.023}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 Highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":128000}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5-highspeed":{"id":"minimax/minimax-m2.5-highspeed","name":"MiniMax M2.5 Highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":8192},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"nvidia/nemotron-nano-9b-v2":{"id":"nvidia/nemotron-nano-9b-v2","name":"Nemotron Nano 9B","description":"Compact Nemotron model for efficient reasoning and deployable AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-18","last_updated":"2025-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.06,"output":0.23}},"nvidia/nemotron-3.5-lightning-30b-a3b":{"id":"nvidia/nemotron-3.5-lightning-30b-a3b","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":262144},"cost":{"input":0,"output":0}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3-7-sonnet-20250219":{"id":"anthropic/claude-3-7-sonnet-20250219","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-1-20250805":{"id":"anthropic/claude-opus-4-1-20250805","name":"Claude Opus 4.1 (20250805)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-20250514":{"id":"anthropic/claude-opus-4-20250514","name":"Claude Opus 4 (20250514)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-5-20250929":{"id":"anthropic/claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5 (20250929)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5-20251001":{"id":"anthropic/claude-haiku-4-5-20251001","name":"Claude Haiku 4.5 (20251001)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":128000}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4-20250514":{"id":"anthropic/claude-sonnet-4-20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5 (20251101)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.13,"output":0.4}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Gemini 2.5 Flash Image","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Gemini 3 Pro Image","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":2,"output":12}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-2.5-computer-use-preview-10-2025":{"id":"google/gemini-2.5-computer-use-preview-10-2025","name":"Gemini 2.5 Computer Use Preview (10-2025)","description":"Specialized Gemini 2.5 model for browser-control agents that automate UI tasks","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":1.25,"output":10}},"google/gemini-3-pro-preview":{"id":"google/gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash-Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-embedding-001":{"id":"google/gemini-embedding-001","name":"Gemini Embedding 001","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"gemini","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-05","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2048,"output":4096},"cost":{"input":0.15,"output":0}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Gemini 3.1 Flash Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash-Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"google/gemini-flash-lite-latest":{"id":"google/gemini-flash-lite-latest","name":"Gemini Flash-Lite Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B It","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.14,"output":0.4}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048000,"output":32000},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B","description":"Compact open Llama model for lightweight chat, drafting, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.22,"output":0.22}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":262144},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":262144},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/llama-3.1-70b-instruct":{"id":"meta/llama-3.1-70b-instruct","name":"Llama 3.1 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.99,"output":0.99}},"meta/llama-3.3-70b-instruct":{"id":"meta/llama-3.3-70b-instruct","name":"Llama 3.3 70B Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.22,"output":0.5,"cache_read":0.11}},"bytedance/dola-seed-2.0-code-preview":{"id":"bytedance/dola-seed-2.0-code-preview","name":"Dola Seed 2.0 Code (preview)","description":"Preview coding model for repository understanding, refactors, and engineering tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-28","last_updated":"2026-03-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"bytedance/dola-seed-2.0-code":{"id":"bytedance/dola-seed-2.0-code","name":"Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.4,"output":2.4}},"bytedance/dola-seed-2.0-lite":{"id":"bytedance/dola-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Efficient Seed model for general chat, analysis, and lightweight production tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-28","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.25,"output":2}},"bytedance/dola-seed-2.0-pro":{"id":"bytedance/dola-seed-2.0-pro","name":"Seed 2.0 Pro","description":"Higher-capability Seed model for complex chat, analysis, and production tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-28","last_updated":"2026-03-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"bytedance/dola-seed-2.0-mini":{"id":"bytedance/dola-seed-2.0-mini","name":"Seed 2.0 Mini","description":"Low-cost Seed model for general chat, extraction, and lightweight production tasks","family":"seed","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.4}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Palmyra X5","description":"Enterprise multimodal model for writing, analysis, and tool-assisted workflows","family":"palmyra","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":0.6,"output":6}},"writer/palmyra-x4":{"id":"writer/palmyra-x4","name":"Palmyra X4","description":"Enterprise language model for writing, analysis, and tool-assisted workflows","family":"palmyra","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-10-09","last_updated":"2024-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":2.5,"output":10}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":250000},"cost":{"input":5,"output":30,"cache_read":0.5}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":2.9,"output":14,"cache_read":0.3}},"moonshot/kimi-k2.5":{"id":"moonshot/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"deepseek/deepseek-v4-flash-0731-fast":{"id":"deepseek/deepseek-v4-flash-0731-fast","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.28,"output":0.56,"cache_read":0.07}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v3":{"id":"deepseek/deepseek-v3","name":"DeepSeek V3","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":81920},"cost":{"input":0.58,"output":1.68}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":40960},"cost":{"input":1.35,"output":5.4}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":40960},"cost":{"input":0.28,"output":0.45,"cache_read":0.14}},"deepseek/deepseek-v4-pro-0423":{"id":"deepseek/deepseek-v4-pro-0423","name":"DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":393216},"cost":{"input":1.65,"output":3.3}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":41000},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 Nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":5}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 Mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5-chat-latest":{"id":"openai/gpt-5-chat-latest","name":"GPT-5 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT-OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.04,"output":0.2,"cache_read":0.02}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-chat-latest":{"id":"openai/gpt-5.1-chat-latest","name":"GPT-5.1 Chat Latest","description":"Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-oss-safeguard-120b":{"id":"openai/gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4096,"output":4096},"cost":{"input":0.15,"output":0.6}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o Mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT-OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.36}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.2-chat-latest":{"id":"openai/gpt-5.2-chat-latest","name":"GPT-5.2 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4 Mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3 Mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.3-chat-latest":{"id":"openai/gpt-5.3-chat-latest","name":"GPT-5.3 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.6,"output":2.5}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+ 08-2024","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a-03-2025":{"id":"cohere/command-a-03-2025","name":"Command A 03-2025","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B 12-2024","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R 08-2024","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 Non-Reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-4.5-air":{"id":"zai/glm-4.5-air","name":"GLM-4.5 Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1,"max":50000}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.05,"output":3.3,"cache_read":0.195}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.015,"output":0.05,"cache_read":0.003}},"zai/glm-4.5":{"id":"zai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"zai/glm-4.5v":{"id":"zai/glm-4.5v","name":"Glm 4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.6,"output":1.8,"cache_read":0.11,"cache_write":0}},"zai/glm-4.7-flashx":{"id":"zai/glm-4.7-flashx","name":"GLM-4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5 Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.7,"output":2.2,"cache_read":0.13}},"zai/glm-4.7-flash":{"id":"zai/glm-4.7-flash","name":"GLM 4.7 Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.07,"output":0.4}},"mistral/devstral-small-2507":{"id":"mistral/devstral-small-2507","name":"Devstral Small","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/magistral-medium-latest":{"id":"mistral/magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2,"output":5}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/devstral-medium-2507":{"id":"mistral/devstral-medium-2507","name":"Devstral Medium","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/devstral-medium-latest":{"id":"mistral/devstral-medium-latest","name":"Devstral 2 (latest)","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"mistral/mistral-large-2411":{"id":"mistral/mistral-large-2411","name":"Mistral Large 2.1","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":2,"output":6}},"mistral/mistral-medium-2505":{"id":"mistral/mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/codestral-latest":{"id":"mistral/codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"mistral/pixtral-large-latest":{"id":"mistral/pixtral-large-latest","name":"Pixtral Large (latest)","description":"Mistral's larger vision model for document-heavy image understanding and chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":2,"output":6}}}},"deepseek":{"id":"deepseek","env":["DEEPSEEK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.deepseek.com","name":"DeepSeek","doc":"https://api-docs.deepseek.com/quick_start/pricing","models":{"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"reasoning":0.87,"cache_read":0.003625}},"deepseek-flash":{"id":"deepseek-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"reasoning":0.6,"cache_read":0.003}}}},"abacus":{"id":"abacus","env":["ABACUS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://routellm.abacus.ai/v1","name":"Abacus","doc":"https://abacus.ai/help/api","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2.5,"output":7.5}},"grok-4-1-fast-non-reasoning":{"id":"grok-4-1-fast-non-reasoning","name":"Grok 4.1 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-11-17","last_updated":"2025-11-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":16384},"cost":{"input":0.2,"output":0.5}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"qwen-2.5-coder-32b":{"id":"qwen-2.5-coder-32b","name":"Qwen 2.5 Coder 32B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.79,"output":0.79}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"grok-code-fast-1":{"id":"grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-09-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":0.2,"output":1.5}},"gemini-3-pro-image":{"id":"gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"claude-3-7-sonnet-20250219":{"id":"claude-3-7-sonnet-20250219","name":"Claude Sonnet 3.7","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10-31","release_date":"2025-02-19","last_updated":"2025-02-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"kimi-k2-turbo-preview":{"id":"kimi-k2-turbo-preview","name":"Kimi K2 Turbo Preview","description":"Fast Kimi model for responsive chat, coding help, and agent loops","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":8192},"cost":{"input":0.15,"output":8}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32768},"cost":{"input":2,"output":6}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"claude-opus-4-20250514":{"id":"claude-opus-4-20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"GPT-5.1 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"qwen3-max":{"id":"qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":1.2,"output":6}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.18}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"gemini-3.1-flash-image":{"id":"gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":3}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"grok-4-0709":{"id":"grok-4-0709","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16384},"cost":{"input":3,"output":15}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.3-codex-xhigh":{"id":"gpt-5.3-codex-xhigh","name":"GPT-5.3 Codex XHigh","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"cache_read":0.2}},"muse-spark-1.1":{"id":"muse-spark-1.1","name":"Muse Spark 1.1","description":"Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.2}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":32768},"cost":{"input":2,"output":6,"cache_read":0.5}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.6,"output":3}},"route-llm":{"id":"route-llm","name":"RouteLLM","description":"RouteLLM routes prompts to an appropriate Abacus-backed text-generation model","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-01-01","last_updated":"2026-07-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":3,"output":15}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"cache_read":0.075}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":1048576,"output":32768},"cost":{"input":0.5,"output":3}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"claude-sonnet-4-20250514":{"id":"claude-sonnet-4-20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-14","last_updated":"2025-05-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"GPT-5.2 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2026-01-01","last_updated":"2026-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"llama-3.3-70b-versatile":{"id":"llama-3.3-70b-versatile","name":"Llama 3.3 70B Versatile","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.59,"output":0.79}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"Grok 4 Fast (Non-Reasoning)","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":16384},"cost":{"input":0.2,"output":0.5}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-pro":{"id":"o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":40}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.75,"output":14}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"gpt-4o-2024-11-20":{"id":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-01","last_updated":"2025-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":3,"output":7}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-15","last_updated":"2025-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.27,"output":0.4}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":1.74,"output":3.48,"cache_read":0.15}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.14,"output":0.4}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-4.5":{"id":"zai-org/GLM-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":96000},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":3.74,"output":9.36,"cache_read":0.748}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-01","last_updated":"2025-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.29,"output":1.2}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen 2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.11,"output":0.38}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.09,"output":0.29}},"Qwen/QwQ-32B":{"id":"Qwen/QwQ-32B","name":"QwQ 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2024-11-28","last_updated":"2024-11-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.4,"output":0.4}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.32,"output":3.2}},"deepseek/deepseek-v3.1":{"id":"deepseek/deepseek-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.55,"output":1.66}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"meta-llama/Meta-Llama-3.1-8B-Instruct":{"id":"meta-llama/Meta-Llama-3.1-8B-Instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.02,"output":0.05}},"meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo":{"id":"meta-llama/Meta-Llama-3.1-405B-Instruct-Turbo","name":"Llama 3.1 405B Instruct Turbo","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":3.5,"output":3.5}},"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8":{"id":"meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":8192},"cost":{"input":0.14,"output":0.59}},"meta-llama/Meta-Llama-3.3-70B-Instruct":{"id":"meta-llama/Meta-Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.59,"output":0.79}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.08,"output":0.44}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"blueclaw":{"id":"blueclaw","env":["BLUECLAW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://openai.blueclaw.network/v1","name":"Blue Claw","doc":"https://blueclaw.network","models":{"Qwen3.6-27B":{"id":"Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"status":"beta"},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"status":"beta"}}},"kosmik":{"id":"kosmik","env":["KOSMIK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.koscompute.com/v1","name":"Kosmik Compute","doc":"https://api.koscompute.com/docs/","models":{"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.35,"output":2.2,"cache_read":0.09}}}},"opencode":{"id":"opencode","env":["OPENCODE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://opencode.ai/zen/v1","name":"OpenCode Zen","doc":"https://opencode.ai/docs/zen","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"nemotron-3-ultra-free":{"id":"nemotron-3-ultra-free","name":"Nemotron 3 Ultra Free","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-02","release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":2.2,"cache_read":0.1}},"hy3-preview-free":{"id":"hy3-preview-free","name":"Hy3 preview Free","description":"Legacy model retained for compatibility with older integrations","family":"hy3-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"grok-code":{"id":"grok-code","name":"Grok Code Fast 1","description":"Legacy model retained for compatibility with older integrations","family":"grok","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-08-20","last_updated":"2025-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"muse-spark-1.3":{"id":"muse-spark-1.3","name":"Muse Spark 1.3","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for long-running agentic, multi-agent, and coding workflows. It improves long-horizon agent collaboration, instruction following, and coding efficiency relative to Muse Spark 1.2.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"muse-spark-1.3-contributor-free":{"id":"muse-spark-1.3-contributor-free","name":"Muse Spark 1.3 Free","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for coding and agentic workflows.","family":"muse-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5-codex":{"id":"gpt-5-codex","name":"GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"muse-spark-1.2":{"id":"muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":2.2,"cache_read":0.1}},"north-mini-code-free":{"id":"north-mini-code-free","name":"North Mini Code Free","description":"Cohere coding model for practical software engineering and agentic edits","family":"north-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-23","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.3,"output":1.2,"cache_read":0.1}},"minimax-m2.1-free":{"id":"minimax-m2.1-free","name":"MiniMax-M2.1 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.1-codex-mini":{"id":"gpt-5.1-codex-mini","name":"GPT-5.1 Codex Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"gpt-5.1-codex":{"id":"gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol (50% Off)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"longcat-2.0-free":{"id":"longcat-2.0-free","name":"LongCat-2.0 Free","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-flash-free":{"id":"deepseek-v4-flash-free","name":"DeepSeek V4 Flash Free","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":128000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"laguna-s-2.1-free":{"id":"laguna-s-2.1-free","name":"Laguna S 2.1 Free","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Legacy model retained for compatibility with older integrations","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2.5,"cache_read":0.4}},"gpt-5.3-codex-spark":{"id":"gpt-5.3-codex-spark","name":"GPT-5.3 Codex Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-spark","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":128000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"minimax-m3-free":{"id":"minimax-m3-free","name":"MiniMax-M3 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-m3-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1,"output":2,"cache_read":0.2}},"qwen3-coder":{"id":"qwen3-coder","name":"Qwen3 Coder","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.45,"output":1.8}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"gpt-5.1-codex-max":{"id":"gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"muse-spark-1.2-contributor-free":{"id":"muse-spark-1.2-contributor-free","name":"Muse Spark 1.2 Free","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0,"output":0,"cache_read":0}},"x-preview-f-free":{"id":"x-preview-f-free","name":"Ox Alpha Free (Unlimited)","description":"Stealth reasoning model for coding, agentic tasks, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"ling-2.6-flash-free":{"id":"ling-2.6-flash-free","name":"Ling 2.6 Flash Free","description":"Legacy model retained for compatibility with older integrations","family":"ling-flash-free","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262100,"output":32800},"status":"deprecated","cost":{"input":0,"output":0}},"gemini-3-pro":{"id":"gemini-3-pro","name":"Gemini 3 Pro","description":"Legacy model retained for compatibility with older integrations","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/google"},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"nemotron-3.5-lightning-free":{"id":"nemotron-3.5-lightning-free","name":"Nemotron 3.5 Lightning Free","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"hy3-free":{"id":"hy3-free","name":"Hy3 Free","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hy3-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":190000,"input":192000,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-24","last_updated":"2026-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"kimi-k2.5-free":{"id":"kimi-k2.5-free","name":"Kimi K2.5 Free","description":"Legacy model retained for compatibility with older integrations","family":"kimi-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"ring-2.6-1t-free":{"id":"ring-2.6-1t-free","name":"Ring 2.6 1T Free","description":"Legacy model retained for compatibility with older integrations","family":"ring-1t-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-06","release_date":"2026-05-08","last_updated":"2026-05-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":66000},"status":"deprecated","cost":{"input":0,"output":0}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"mimo-v2.5-free":{"id":"mimo-v2.5-free","name":"MiMo V2.5 Free","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo-v2.5-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"claude-3-5-haiku":{"id":"claude-3-5-haiku","name":"Claude Haiku 3.5","description":"Legacy model retained for compatibility with older integrations","family":"claude-haiku","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07-31","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"nemotron-3-super-free":{"id":"nemotron-3-super-free","name":"Nemotron 3 Super Free","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-02","release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180,"cache_read":30}},"big-pickle":{"id":"big-pickle","name":"Big Pickle","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"big-pickle","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-17","last_updated":"2025-10-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":160000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"ling-3.0-flash-fin-free":{"id":"ling-3.0-flash-fin-free","name":"Ling 3.0 Flash Fin Free","description":"Finance-enhanced model for financial research, multi-step investment workflows, and long-horizon planning and execution","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2":{"id":"kimi-k2","name":"Kimi K2","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2.5,"cache_read":0.4}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.6,"output":3,"cache_read":0.08}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"ling-3.0-flash-free":{"id":"ling-3.0-flash-free","name":"Ling-3.0-flash Free","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"trinity-large-preview-free":{"id":"trinity-large-preview-free","name":"Trinity Large Preview","description":"Legacy model retained for compatibility with older integrations","family":"trinity","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-01-27","last_updated":"2026-01-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0,"output":0}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.84,"cache_read":0.145}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":30,"output":180,"cache_read":30}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"glm-4.7-free":{"id":"glm-4.7-free","name":"GLM-4.7 Free","description":"Legacy model retained for compatibility with older integrations","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"glm-5-free":{"id":"glm-5-free","name":"GLM-5 Free","description":"Legacy model retained for compatibility with older integrations","family":"glm-free","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"mimo-v2-flash-free":{"id":"mimo-v2-flash-free","name":"MiMo V2 Flash Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-flash-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google"},"cost":{"input":0.5,"output":3,"cache_read":0.05}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":1.07,"output":8.5,"cache_read":0.107}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"minimax-m2.5-free":{"id":"minimax-m2.5-free","name":"MiniMax-M2.5 Free","description":"Legacy model retained for compatibility with older integrations","family":"minimax-free","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-omni-free":{"id":"mimo-v2-omni-free","name":"MiMo V2 Omni Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-omni-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-pro-free":{"id":"mimo-v2-pro-free","name":"MiMo V2 Pro Free","description":"Legacy model retained for compatibility with older integrations","family":"mimo-pro-free","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"qwen3.6-plus-free":{"id":"qwen3.6-plus-free","name":"Qwen3.6 Plus Free","description":"Legacy model retained for compatibility with older integrations","family":"qwen-free","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0,"output":0,"cache_read":0}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"ling-3.0-tiny-free":{"id":"ling-3.0-tiny-free","name":"Ling-3.0-tiny Free","description":"Compact MoE model for responsive agents, instruction following, and multi-turn conversations","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"status":"deprecated","cost":{"input":0,"output":0}}}},"moonshotai-cn":{"id":"moonshotai-cn","env":["MOONSHOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.moonshot.cn/v1","name":"Moonshot AI (China)","doc":"https://platform.moonshot.cn/docs/api/chat","models":{"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code HighSpeed","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}}}},"stepfun-step-plan":{"id":"stepfun-step-plan","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.com/step_plan/v1","name":"StepFun Step Plan (China)","doc":"https://platform.stepfun.com/docs/zh/step-plan/integrations/reasoning-api","models":{"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-router-v1":{"id":"step-router-v1","name":"Step Router v1","description":"StepFun routing model that dispatches requests to the appropriate Step model.","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"input":256000,"output":256000}}}},"nearai":{"id":"nearai","env":["NEARAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://cloud-api.near.ai/v1","name":"NEAR AI Cloud","doc":"https://docs.near.ai/","models":{"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"input_audio":0.3}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"input_audio":1}},"zai-org/GLM-5.1-FP8":{"id":"zai-org/GLM-5.1-FP8","name":"GLM-5.1 FP8","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":16384},"cost":{"input":1.4,"output":4.4}},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen 3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":8192},"cost":{"input":0.17,"output":1.1,"cache_read":0.056}},"Qwen/Qwen3-Embedding-0.6B":{"id":"Qwen/Qwen3-Embedding-0.6B","name":"Qwen3 Embedding 0.6B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":1024},"cost":{"input":0.01,"output":0.01}},"Qwen/Qwen3-Reranker-0.6B":{"id":"Qwen/Qwen3-Reranker-0.6B","name":"Qwen3 Reranker 0.6B","description":"Reranking model for improving retrieval quality in search and recommendation systems","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-06-03","last_updated":"2025-06-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":40960,"output":1024},"cost":{"input":0.01,"output":0.01}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen3-VL 30B-A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":8192},"cost":{"input":0.15,"output":0.55}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/whisper-large-v3":{"id":"openai/whisper-large-v3","name":"Whisper Large v3","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"whisper","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-11-06","last_updated":"2023-11-06","modalities":{"input":["audio"],"output":["text"]},"open_weights":true,"limit":{"context":448,"output":448},"cost":{"input":0.01,"output":0.01}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"black-forest-labs/FLUX.2-klein-4B":{"id":"black-forest-labs/FLUX.2-klein-4B","name":"FLUX.2 Klein 4B","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"flux","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-01-14","last_updated":"2026-01-14","modalities":{"input":["text"],"output":["image"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":1,"output":1}}}},"openrouter":{"id":"openrouter","env":["OPENROUTER_API_KEY"],"npm":"@openrouter/ai-sdk-provider","api":"https://openrouter.ai/api/v1","name":"OpenRouter","doc":"https://openrouter.ai/models","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.475,"output":4.425,"cache_read":0.295,"cache_write":1.84375}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.65,"output":3.25,"cache_read":0.13,"cache_write":0.8125,"tiers":[{"input":1.17,"output":5.85,"cache_read":0.234,"cache_write":1.4625,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"cache_read":0.39,"cache_write":2.4375,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-235b-a22b-thinking-2507":{"id":"qwen/qwen3-235b-a22b-thinking-2507","name":"Qwen3 235B A22B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.23,"output":2.3}},"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.1,"output":0.15}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":1.1}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.195,"output":0.975,"cache_read":0.039,"cache_write":0.24375,"tiers":[{"input":0.325,"output":1.625,"cache_read":0.065,"cache_write":0.40625,"tier":{"type":"context","size":32000}},{"input":0.52,"output":2.6,"cache_read":0.104,"cache_write":0.65,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-14b":{"id":"qwen/qwen3-14b","name":"Qwen3 14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.2275,"output":0.91}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.325,"output":1.95,"cache_write":0.40625,"tiers":[{"input":1.3,"output":3.9,"cache_write":1.625,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.3,"output":3.9,"cache_write":1.625}}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.195,"output":1.56}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.42,"output":3,"cache_read":0.085}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.3125,"output":1.25,"cache_read":0.15625}},"qwen/qwen3.5-plus-20260420":{"id":"qwen/qwen3.5-plus-20260420","name":"Qwen3.5 Plus 2026-04-20","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8,"cache_write":0.375,"tiers":[{"input":0.375,"output":2.25,"cache_write":0.46875,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.375,"output":2.25,"cache_write":0.46875}}},"qwen/qwen3-32b":{"id":"qwen/qwen3-32b","name":"Qwen3 32B","description":"Dense open Qwen model for self-hosted chat, reasoning, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.08,"output":0.28}},"qwen/qwen3.5-plus-02-15":{"id":"qwen/qwen3.5-plus-02-15","name":"Qwen3.5 Plus 2026-02-15","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.26,"output":1.56,"tiers":[{"input":0.325,"output":1.95,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.325,"output":1.95}}},"qwen/qwen-plus-2025-07-28":{"id":"qwen/qwen-plus-2025-07-28","name":"Qwen Plus 0728","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-08","last_updated":"2025-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"tiers":[{"input":0.78,"output":2.34,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.78,"output":2.34}}},"qwen/qwen3-coder":{"id":"qwen/qwen3-coder","name":"Qwen3 Coder 480B A35B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1,"cache_read":0.1}},"qwen/qwen2.5-vl-72b-instruct":{"id":"qwen/qwen2.5-vl-72b-instruct","name":"Qwen2.5 VL 72B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-02-01","last_updated":"2025-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":115200},"cost":{"input":0.8,"output":1,"cache_read":0.4}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.12,"output":0.8,"cache_read":0.07}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.07,"output":0.28}},"qwen/qwen3-235b-a22b-2507":{"id":"qwen/qwen3-235b-a22b-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.22,"output":0.88}},"qwen/qwen3.5-flash-02-23":{"id":"qwen/qwen3.5-flash-02-23","name":"Qwen3.5-Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.065,"output":0.26}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.55,"output":3.5,"cache_read":0.225}},"qwen/qwen3-vl-8b-thinking":{"id":"qwen/qwen3-vl-8b-thinking","name":"Qwen3 VL 8B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.18,"output":2.1}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":2,"cache_read":0.03}},"qwen/qwen3.7-flash":{"id":"qwen/qwen3.7-flash","name":"Qwen3.7 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"input":991000,"output":65536},"cost":{"input":0.03,"output":0.13,"cache_read":0.006,"cache_write":0.038,"tiers":[{"input":0.1,"output":0.4,"cache_read":0.02,"cache_write":0.125,"tier":{"type":"context","size":32000}},{"input":0.2,"output":0.8,"cache_read":0.04,"cache_write":0.25,"tier":{"type":"context","size":256000}}]}},"qwen/qwen3-30b-a3b-thinking-2507":{"id":"qwen/qwen3-30b-a3b-thinking-2507","name":"Qwen3 30B A3B Thinking 2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":81920,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-35b-a3b":{"id":"qwen/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.1,"output":0.9,"cache_read":0.05}},"qwen/qwen3-max-thinking":{"id":"qwen/qwen3-max-thinking","name":"Qwen3 Max Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-09","last_updated":"2026-02-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"tiers":[{"input":1.56,"output":7.8,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.78,"output":3.9,"cache_read":0.156,"cache_write":0.975,"tiers":[{"input":1.56,"output":7.8,"cache_read":0.312,"cache_write":1.95,"tier":{"type":"context","size":32000}},{"input":1.95,"output":9.75,"cache_read":0.39,"cache_write":2.4375,"tier":{"type":"context","size":128000}}]}},"qwen/qwen3-vl-8b-instruct":{"id":"qwen/qwen3-vl-8b-instruct","name":"Qwen3 VL 8B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25}},"qwen/qwen3-vl-30b-a3b-instruct":{"id":"qwen/qwen3-vl-30b-a3b-instruct","name":"Qwen3 VL 30B A3B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.15,"output":0.6}},"qwen/qwen-plus":{"id":"qwen/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.26,"output":0.78,"cache_read":0.052,"cache_write":0.325,"tiers":[{"input":0.78,"output":2.34,"cache_read":0.156,"cache_write":0.975,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.78,"output":2.34,"cache_read":0.156,"cache_write":0.975}}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.26,"output":2.08}},"qwen/qwen-2.5-coder-32b-instruct":{"id":"qwen/qwen-2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-11-11","last_updated":"2024-11-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.66,"output":1}},"qwen/qwen3-30b-a3b-instruct-2507":{"id":"qwen/qwen3-30b-a3b-instruct-2507","name":"Qwen3 30B A3B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.04815,"output":0.19305}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1875,"output":1.125,"cache_write":0.234375,"tiers":[{"input":0.75,"output":3,"cache_write":0.9375,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.75,"output":3,"cache_write":0.9375}}},"qwen/qwen3-30b-a3b":{"id":"qwen/qwen3-30b-a3b","name":"Qwen3 30B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.12,"output":0.5}},"qwen/qwen-2.5-7b-instruct":{"id":"qwen/qwen-2.5-7b-instruct","name":"Qwen2.5 7B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-10-16","last_updated":"2024-10-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.1,"output":0.2}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3-vl-30b-a3b-thinking":{"id":"qwen/qwen3-vl-30b-a3b-thinking","name":"Qwen3 VL 30B A3B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":2.4}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.027,"output":6.162,"cache_write":1.28375,"tiers":[{"input":1.58,"output":9.48,"cache_write":1.975,"tier":{"type":"context","size":128000}}]}},"qwen/qwen-2.5-72b-instruct":{"id":"qwen/qwen-2.5-72b-instruct","name":"Qwen2.5 72B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.36,"output":0.4}},"qwen/qwen3-8b":{"id":"qwen/qwen3-8b","name":"Qwen3 8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.117,"output":0.455}},"qwen/qwen3-235b-a22b":{"id":"qwen/qwen3-235b-a22b","name":"Qwen3 235B-A22B","description":"Large open Qwen MoE for multilingual reasoning, coding, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.455,"output":1.82}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.21,"output":1.9,"cache_read":0.1}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.32,"output":1.28,"cache_read":0.064,"cache_write":0.4,"tiers":[{"input":0.96,"output":3.84,"cache_read":0.192,"cache_write":1.2,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.96,"output":3.84,"cache_read":0.192,"cache_write":1.2}}},"qwen/qwen3-vl-32b-instruct":{"id":"qwen/qwen3-vl-32b-instruct","name":"Qwen3 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-23","last_updated":"2025-10-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.104,"output":0.416}},"baidu/ernie-4.5-vl-424b-a47b":{"id":"baidu/ernie-4.5-vl-424b-a47b","name":"ERNIE 4.5 VL 424B A47B ","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ernie","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-06-30","last_updated":"2025-06-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":123000,"output":16000},"cost":{"input":0.42,"output":1.25}},"aion-labs/aion-2.0":{"id":"aion-labs/aion-2.0","name":"Aion-2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.8,"output":1.6,"cache_read":0.2}},"aion-labs/aion-rp-llama-3.1-8b":{"id":"aion-labs/aion-rp-llama-3.1-8b","name":"Aion-RP 1.0 (8B)","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12-31","release_date":"2025-02-04","last_updated":"2025-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":29491},"cost":{"input":0.8,"output":1.6}},"aion-labs/aion-3.0":{"id":"aion-labs/aion-3.0","name":"Aion-3.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":3,"output":6,"cache_read":0.75}},"aion-labs/aion-3.0-mini":{"id":"aion-labs/aion-3.0-mini","name":"Aion-3.0-Mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-07","last_updated":"2026-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.7,"output":1.4,"cache_read":0.18}},"~anthropic/claude-fable-latest":{"id":"~anthropic/claude-fable-latest","name":"Claude Fable Latest","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"~anthropic/claude-opus-latest":{"id":"~anthropic/claude-opus-latest","name":"Claude Opus Latest","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"~anthropic/claude-haiku-latest":{"id":"~anthropic/claude-haiku-latest","name":"Anthropic Claude Haiku Latest","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"~anthropic/claude-sonnet-latest":{"id":"~anthropic/claude-sonnet-latest","name":"Anthropic Claude Sonnet Latest","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"morph/morph-v3-large":{"id":"morph/morph-v3-large","name":"Morph V3 Large","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.9,"output":1.9}},"morph/morph-v3-fast":{"id":"morph/morph-v3-fast","name":"Morph V3 Fast","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"morph","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-07-07","last_updated":"2025-07-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":81920,"output":38000},"cost":{"input":0.8,"output":1.2}},"undi95/remm-slerp-l2-13b":{"id":"undi95/remm-slerp-l2-13b","name":"ReMM SLERP 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-07-22","last_updated":"2023-07-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":6144,"output":5529},"cost":{"input":0.35,"output":0.65}},"~deepseek/deepseek-v4-flash-latest":{"id":"~deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-01","last_updated":"2026-08-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":393216},"cost":{"input":0.05,"output":0.16,"cache_read":0.013}},"dots-studio/dots-3-note-preview:free":{"id":"dots-studio/dots-3-note-preview:free","name":"Dots3-Note Preview (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":460800},"cost":{"input":0,"output":0}},"~x-ai/grok-latest":{"id":"~x-ai/grok-latest","name":"Grok Latest","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"meituan/longcat-2.0":{"id":"meituan/longcat-2.0","name":"LongCat 2.0","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-20","last_updated":"2026-07-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048756,"output":262144},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"poolside/laguna-xs-2.1":{"id":"poolside/laguna-xs-2.1","name":"Laguna XS 2.1","description":"Agentic coding model from Poolside in the XS size class for local deployment","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.06,"output":0.12,"cache_read":0.03}},"poolside/laguna-xs-2.1:free":{"id":"poolside/laguna-xs-2.1:free","name":"Laguna XS 2.1 (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-02","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1:free":{"id":"poolside/laguna-s-2.1:free","name":"Laguna S 2.1 (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"poolside/laguna-s-2.1":{"id":"poolside/laguna-s-2.1","name":"Laguna S 2.1","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"laguna-s","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.09,"output":0.18,"cache_read":0.009}},"kwaipilot/kat-coder-pro-v2":{"id":"kwaipilot/kat-coder-pro-v2","name":"KAT-Coder-Pro V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-27","last_updated":"2026-03-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":144000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kwaipilot/kat-coder-pro-v2.5":{"id":"kwaipilot/kat-coder-pro-v2.5","name":"KAT-Coder-Pro V2.5","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"kat-coder","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-10","last_updated":"2026-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.74,"output":2.96,"cache_read":0.15}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":256000,"output":230400},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3}},"mistralai/ministral-14b-2512":{"id":"mistralai/ministral-14b-2512","name":"Ministral 3 14B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.2,"output":0.2,"cache_read":0.02}},"mistralai/mistral-large":{"id":"mistralai/mistral-large","name":"Mistral Large","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11-30","release_date":"2024-02-26","last_updated":"2024-02-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":102400},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/codestral-2508":{"id":"mistralai/codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"codestral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-01","last_updated":"2025-08-01","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":204800},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"mistralai/mistral-medium-3-5":{"id":"mistralai/mistral-medium-3-5","name":"Mistral Medium 3.5","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":209715},"cost":{"input":1.5,"output":7.5}},"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-large-2407":{"id":"mistralai/mistral-large-2407","name":"Mistral Large 2407","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-03-31","release_date":"2024-11-19","last_updated":"2024-11-19","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-small-3.2-24b-instruct":{"id":"mistralai/mistral-small-3.2-24b-instruct","name":"Mistral Small 3.2 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.075,"output":0.2}},"mistralai/mixtral-8x22b-instruct":{"id":"mistralai/mixtral-8x22b-instruct","name":"Mixtral 8x22B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-01-31","release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":52428},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistralai/mistral-saba":{"id":"mistralai/mistral-saba","name":"Saba","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-02-17","last_updated":"2025-02-17","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":26214},"cost":{"input":0.2,"output":0.6,"cache_read":0.02}},"mistralai/mistral-large-2512":{"id":"mistralai/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistralai/ministral-3b-2512":{"id":"mistralai/ministral-3b-2512","name":"Ministral 3 3B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":104857},"cost":{"input":0.1,"output":0.1,"cache_read":0.01}},"mistralai/mistral-nemo":{"id":"mistralai/mistral-nemo","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.019,"output":0.03}},"mistralai/mistral-medium-3":{"id":"mistralai/mistral-medium-3","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistralai/mistral-small-24b-instruct-2501":{"id":"mistralai/mistral-small-24b-instruct-2501","name":"Mistral Small 3","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-01-30","last_updated":"2025-01-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":16384},"cost":{"input":0.05,"output":0.08}},"mistralai/mistral-small-3.1-24b-instruct":{"id":"mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-10-31","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":102400},"cost":{"input":0.351,"output":0.555}},"mistralai/mistral-small-2603":{"id":"mistralai/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistralai/ministral-8b-2512":{"id":"mistralai/ministral-8b-2512","name":"Ministral 3 8B 2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":209715},"cost":{"input":0.15,"output":0.15,"cache_read":0.015}},"mistralai/voxtral-small-24b-2507":{"id":"mistralai/voxtral-small-24b-2507","name":"Voxtral Small 24B 2507","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":26214},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"mistralai/mistral-medium-3.1":{"id":"mistralai/mistral-medium-3.1","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06-30","release_date":"2025-08-13","last_updated":"2025-08-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":104857},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1050000,"output":131072},"cost":{"input":0.435,"output":0.87,"cache_read":0.0036}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":false,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.255,"output":1.02}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.27,"output":1.08,"cache_read":0.027}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/minimax-m2-her":{"id":"minimax/minimax-m2-her","name":"MiniMax-M2 Her","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/minimax-m1":{"id":"minimax/minimax-m1","name":"MiniMax M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":40000},"cost":{"input":0.55,"output":2.2}},"minimax/minimax-01":{"id":"minimax/minimax-01","name":"MiniMax-01","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-03-31","release_date":"2025-01-15","last_updated":"2025-01-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000192,"output":900172},"cost":{"input":0.2,"output":1.1}},"nvidia/nemotron-3.5-lightning:free":{"id":"nvidia/nemotron-3.5-lightning:free","name":"Nemotron 3.5 Lightning (free)","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety":{"id":"nvidia/nemotron-3.5-content-safety","name":"Nemotron 3.5 Content Safety","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.2,"output":0.2}},"nvidia/nemotron-3.5-lightning":{"id":"nvidia/nemotron-3.5-lightning","name":"Nemotron 3.5 Lightning 30B A3B","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.08,"output":0.2,"cache_read":0.04}},"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free":{"id":"nvidia/nemotron-3-nano-omni-30b-a3b-reasoning:free","name":"Nemotron 3 Nano Omni (free)","description":"Open Nemotron omni model combining reasoning with text, vision, and audio","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-28","last_updated":"2026-04-28","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b":{"id":"nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.085,"output":0.4}},"nvidia/nemotron-3-ultra-550b-a55b:free":{"id":"nvidia/nemotron-3-ultra-550b-a55b:free","name":"Nemotron 3 Ultra (free)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-super-120b-a12b:free":{"id":"nvidia/nemotron-3-super-120b-a12b:free","name":"Nemotron 3 Super (free)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nvidia/nemotron-3.5-content-safety:free":{"id":"nvidia/nemotron-3.5-content-safety:free","name":"Nemotron 3.5 Content Safety (free)","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"nemotron","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["medium","high"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.625,"output":3.125,"cache_read":0.1875}},"nvidia/nemotron-3-nano-30b-a3b":{"id":"nvidia/nemotron-3-nano-30b-a3b","name":"Nemotron 3 Nano 30B A3B","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.05,"output":0.2,"cache_read":0.03}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-3-haiku":{"id":"anthropic/claude-3-haiku","name":"Claude 3 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-08-31","release_date":"2024-03-13","last_updated":"2024-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":4096},"cost":{"input":0.25,"output":1.25,"cache_read":0.03,"cache_write":0.3}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemma-4-26b-a4b-it":{"id":"google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.042,"output":0.22}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.25,"output":1.5}},"google/gemma-3-4b-it":{"id":"google/gemma-3-4b-it","name":"Gemma 3 4B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.1}},"google/lyria-3-clip-preview":{"id":"google/lyria-3-clip-preview","name":"Lyria 3 Clip Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.1,"output":0.4,"reasoning":0.4,"cache_read":0.01,"cache_write":0.083333}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333}},"google/gemma-3-27b-it":{"id":"google/gemma-3-27b-it","name":"Gemma 3 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.08,"output":0.45,"cache_read":0.04}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-2.5-pro-preview":{"id":"google/gemini-2.5-pro-preview","name":"Gemini 2.5 Pro Preview 06-05","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-06-05","last_updated":"2025-06-05","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375}},"google/gemma-4-31b-it:free":{"id":"google/gemma-4-31b-it:free","name":"Gemma 4 31B (free)","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0.09,"output":0.34,"cache_read":0.05}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/lyria-3-pro-preview":{"id":"google/lyria-3-pro-preview","name":"Lyria 3 Pro Preview","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"lyria","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-25","last_updated":"2026-03-25","modalities":{"input":["text","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0,"output":0}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":58982},"cost":{"input":0.5,"output":3}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333}},"google/gemini-2.5-pro-preview-05-06":{"id":"google/gemini-2.5-pro-preview-05-06","name":"Gemini 2.5 Pro Preview 05-06","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image","pdf","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.25,"output":10,"reasoning":10,"cache_read":0.125,"cache_write":0.375,"tiers":[{"input":2.5,"output":15,"cache_read":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":15,"cache_read":0.25}}},"google/gemma-3-12b-it":{"id":"google/gemma-3-12b-it","name":"Gemma 3 12B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.05,"output":0.15}},"google/gemma-4-26b-a4b-it:free":{"id":"google/gemma-4-26b-a4b-it:free","name":"Gemma 4 26B A4B  (free)","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"google/gemma-2-27b-it":{"id":"google/gemma-2-27b-it","name":"Gemma 2 27B","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-07-13","last_updated":"2024-07-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":2048},"cost":{"input":0.65,"output":0.65}},"relace/relace-apply-3":{"id":"relace/relace-apply-3","name":"Relace Apply 3","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.85,"output":1.25}},"relace/relace-search":{"id":"relace/relace-search","name":"Relace Search","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":1,"output":3}},"nex-agi/nex-n2.5-mini:free":{"id":"nex-agi/nex-n2.5-mini:free","name":"Nex-N2.5-Mini (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"nex-agi/nex-n2.5-pro:free":{"id":"nex-agi/nex-n2.5-pro:free","name":"Nex-N2.5-Pro (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"agi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0,"output":0}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"thinkingmachines/inkling-small:free":{"id":"thinkingmachines/inkling-small:free","name":"Inkling Small (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling:free":{"id":"thinkingmachines/inkling:free","name":"Inkling (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":0,"output":0}},"thinkingmachines/inkling":{"id":"thinkingmachines/inkling","name":"Inkling","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"gryphe/mythomax-l2-13b":{"id":"gryphe/mythomax-l2-13b","name":"MythoMax 13B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-07-02","last_updated":"2023-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":3686},"cost":{"input":0.06,"output":0.06}},"meta/muse-spark-1.3":{"id":"meta/muse-spark-1.3","name":"Muse Spark 1.3","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2":{"id":"meta/muse-spark-1.2","name":"Muse Spark 1.2","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-spark-1.2-contributor":{"id":"meta/muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.3-contributor":{"id":"meta/muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"meta/muse-spark-1.1":{"id":"meta/muse-spark-1.1","name":"Muse Spark 1.1","description":"Open Llama multimodal model for image understanding and text reasoning","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-07-09","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":1.25,"output":4.25,"cache_read":0.15}},"meta/muse-glimmer-30b":{"id":"meta/muse-glimmer-30b","name":"Muse Glimmer 30B","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":1.1,"cache_read":0.04}},"perceptron/perceptron-mk1":{"id":"perceptron/perceptron-mk1","name":"Perceptron Mk1","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-05-12","last_updated":"2026-05-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.15,"output":1.5}},"thedrummer/skyfall-36b-v2":{"id":"thedrummer/skyfall-36b-v2","name":"Skyfall 36B V2","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-03-10","last_updated":"2025-03-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":29491},"cost":{"input":0.55,"output":0.8,"cache_read":0.25}},"thedrummer/unslopnemo-12b":{"id":"thedrummer/unslopnemo-12b","name":"UnslopNemo 12B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-11-08","last_updated":"2024-11-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":819200},"cost":{"input":0.4,"output":0.4}},"thedrummer/cydonia-24b-v4.1":{"id":"thedrummer/cydonia-24b-v4.1","name":"Cydonia 24B V4.1","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04-30","release_date":"2025-09-27","last_updated":"2025-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.3,"output":0.5,"cache_read":0.15}},"bytedance/ui-tars-1.5-7b":{"id":"bytedance/ui-tars-1.5-7b","name":"UI-TARS 7B ","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":2048},"cost":{"input":0.1,"output":0.2,"cache_read":0.1}},"bytedance-seed/seed-1.6-flash":{"id":"bytedance-seed/seed-1.6-flash","name":"Seed 1.6 Flash","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.075,"output":0.3,"tiers":[{"input":0.1,"output":0.8,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2-1-turbo":{"id":"bytedance-seed/seed-2-1-turbo","name":"Seed 2.1 Turbo","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.5,"output":2.5}},"bytedance-seed/seed-2.0-code":{"id":"bytedance-seed/seed-2.0-code","name":"Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.5,"output":3,"tiers":[{"input":1,"output":6,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-1.6":{"id":"bytedance-seed/seed-1.6","name":"Seed 1.6","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0.25,"output":2,"tiers":[{"input":0.5,"output":4,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2.0-mini":{"id":"bytedance-seed/seed-2.0-mini","name":"Seed 2.0 Mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.1,"output":0.4,"tiers":[{"input":0.2,"output":0.8,"tier":{"type":"context","size":128000}}]}},"bytedance-seed/seed-2.0-lite":{"id":"bytedance-seed/seed-2.0-lite","name":"Seed 2.0 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0.25,"output":2,"tiers":[{"input":0.5,"output":4,"tier":{"type":"context","size":128000}}]}},"inception/mercury-2.5":{"id":"inception/mercury-2.5","name":"Mercury 2.5","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-08","last_updated":"2026-09-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":260000,"output":65536},"cost":{"input":0.04,"output":0.15,"cache_read":0.004}},"inception/mercury-2":{"id":"inception/mercury-2","name":"Mercury 2","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"mercury","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-04","last_updated":"2026-03-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":50000},"cost":{"input":0.25,"output":0.75,"cache_read":0.025}},"writer/palmyra-x5":{"id":"writer/palmyra-x5","name":"Palmyra X5","description":"General-purpose chat model for instruction following, writing, and analysis","family":"palmyra","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-01-21","last_updated":"2026-01-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1040000,"output":8192},"cost":{"input":0.6,"output":6}},"~google/gemini-pro-latest":{"id":"~google/gemini-pro-latest","name":"Google Gemini Pro Latest","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["audio","pdf","image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"~google/gemini-flash-latest":{"id":"~google/gemini-flash-latest","name":"Google Gemini Flash Latest","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667}},"microsoft/phi-4":{"id":"microsoft/phi-4","name":"Phi 4","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"phi","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2025-01-10","last_updated":"2025-01-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":14745},"cost":{"input":0.07,"output":0.14}},"microsoft/wizardlm-2-8x22b":{"id":"microsoft/wizardlm-2-8x22b","name":"WizardLM-2 8x22B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04-30","release_date":"2024-04-16","last_updated":"2024-04-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65535,"output":8000},"cost":{"input":0.62,"output":0.62}},"sakana/fugu-ultra":{"id":"sakana/fugu-ultra","name":"Fugu Ultra","description":"Quality-first multi-agent model for hard research, analysis, and competitions","family":"fugu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-15","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"sakana/sakana-namazu":{"id":"sakana/sakana-namazu","name":"Sakana Namazu","description":"Multi-agent model for routing expert agents across complex analytical tasks","family":"sakana-namazu","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.15}},"~moonshotai/kimi-latest":{"id":"~moonshotai/kimi-latest","name":"MoonshotAI Kimi Latest","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":2.4,"output":12,"cache_read":0.24}},"ibm-granite/granite-4.2-8b":{"id":"ibm-granite/granite-4.2-8b","name":"Granite 4.2 8B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"granite","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-31","last_updated":"2026-08-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.06,"output":0.25,"cache_read":0.015}},"ibm-granite/granite-4.0-h-micro":{"id":"ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-10-20","last_updated":"2025-10-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":117900},"cost":{"input":0.017,"output":0.112}},"deepseek/deepseek-chat-v3.1":{"id":"deepseek/deepseek-chat-v3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.25,"output":0.95,"cache_read":0.13}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":943718},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":393216},"cost":{"input":0.57948,"output":1.73844,"cache_read":0.018438}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":943718},"cost":{"input":0.065,"output":0.18,"cache_read":0.016}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.08246,"output":0.16492,"cache_read":0.016492}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"Fast DeepSeek model for efficient chat, coding help, and agent loops","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16000},"cost":{"input":0.7,"output":2.5}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16000},"cost":{"input":0.2574,"output":1.0287}},"deepseek/deepseek-r1-0528":{"id":"deepseek/deepseek-r1-0528","name":"R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.5,"output":2.15,"cache_read":0.35}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.269,"output":0.4,"cache_read":0.1345}},"deepseek/deepseek-r1-distill-llama-70b":{"id":"deepseek/deepseek-r1-distill-llama-70b","name":"R1 Distill Llama 70B","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07-31","release_date":"2025-01-23","last_updated":"2025-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":7372},"cost":{"input":0.8,"output":0.8}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":65536},"cost":{"input":0.27,"output":0.41}},"deepseek/deepseek-v3.1-terminus":{"id":"deepseek/deepseek-v3.1-terminus","name":"DeepSeek V3.1 Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-22","last_updated":"2025-09-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.27,"output":1,"cache_read":0.135}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.87,"output":1.74,"cache_read":0.0725}},"deepseek/deepseek-chat-v3-0324":{"id":"deepseek/deepseek-chat-v3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07-31","release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":147456},"cost":{"input":0.29,"output":1.14,"cache_read":0.11}},"~openai/gpt-latest":{"id":"~openai/gpt-latest","name":"OpenAI GPT Latest","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"~openai/gpt-mini-latest":{"id":"~openai/gpt-mini-latest","name":"OpenAI GPT Mini Latest","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"amazon/nova-2-lite-v1":{"id":"amazon/nova-2-lite-v1","name":"Nova 2 Lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.3,"output":2.5}},"amazon/nova-micro-v1":{"id":"amazon/nova-micro-v1","name":"Nova Micro 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":5120},"cost":{"input":0.035,"output":0.14}},"amazon/nova-pro-v1":{"id":"amazon/nova-pro-v1","name":"Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.8,"output":3.2}},"amazon/nova-premier-v1":{"id":"amazon/nova-premier-v1","name":"Nova Premier 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32000},"cost":{"input":2.5,"output":12.5,"cache_read":0.625}},"amazon/nova-lite-v1":{"id":"amazon/nova-lite-v1","name":"Nova Lite 1.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-10-31","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":5120},"cost":{"input":0.06,"output":0.24}},"inclusionai/ling-3.0-flash":{"id":"inclusionai/ling-3.0-flash","name":"Ling 3.0 Flash","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-23","last_updated":"2026-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.021,"output":0.063,"cache_read":0.0042}},"inclusionai/ling-3.0-flash-fin":{"id":"inclusionai/ling-3.0-flash-fin","name":"Ling 3.0 Flash Fin","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":235929},"cost":{"input":0.06,"output":0.18,"cache_read":0.012}},"inclusionai/ling-3.0-flash-fin:free":{"id":"inclusionai/ling-3.0-flash-fin:free","name":"Ling 3.0 Flash Fin (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-sante:free":{"id":"inclusionai/ling-3.0-flash-sante:free","name":"Ling 3.0 Flash Sante (free)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"inclusionai/ling-3.0-flash-vl:free":{"id":"inclusionai/ling-3.0-flash-vl:free","name":"Ling 3.0 Flash VL (free)","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"anthracite-org/magnum-v4-72b":{"id":"anthracite-org/magnum-v4-72b","name":"Magnum v4 72B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-30","release_date":"2024-10-22","last_updated":"2024-10-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":4096},"cost":{"input":2.5,"output":5}},"mancer/weaver":{"id":"mancer/weaver","name":"Weaver (alpha)","description":"General-purpose chat model for instruction following, writing, and analysis","family":"alpha","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-06-30","release_date":"2023-08-02","last_updated":"2023-08-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":6000},"cost":{"input":0.4,"output":0.75}},"openrouter/free":{"id":"openrouter/free","name":"Free Models Router","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"input":200000,"output":8000},"cost":{"input":0,"output":0}},"openrouter/pareto-code":{"id":"openrouter/pareto-code","name":"Pareto Code Router","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":200000}},"openrouter/bodybuilder":{"id":"openrouter/bodybuilder","name":"Body Builder (beta)","description":"Preview model for early access evaluation, prototyping, and compatibility testing","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000}},"openrouter/fusion":{"id":"openrouter/fusion","name":"Fusion","description":"General-purpose chat model for instruction following, writing, and analysis","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openrouter/auto":{"id":"openrouter/auto","name":"Auto Router","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"auto","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2023-11-08","last_updated":"2023-11-08","modalities":{"input":["text","image","audio","pdf","video"],"output":["text","image"]},"open_weights":false,"limit":{"context":2000000,"output":2000000}},"sao10k/l3.3-euryale-70b":{"id":"sao10k/l3.3-euryale-70b","name":"Llama 3.3 Euryale 70B","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-12-18","last_updated":"2024-12-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.65,"output":0.75}},"sao10k/l3-lunaris-8b":{"id":"sao10k/l3-lunaris-8b","name":"Llama 3 8B Lunaris","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":7372},"cost":{"input":0.04,"output":0.05}},"sao10k/l3.1-euryale-70b":{"id":"sao10k/l3.1-euryale-70b","name":"Llama 3.1 Euryale 70B v2.2","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-28","last_updated":"2024-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.85,"output":0.85}},"x-ai/grok-4.20-multi-agent":{"id":"x-ai/grok-4.20-multi-agent","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":900000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09-01","release_date":"2026-03-31","last_updated":"2026-03-31","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":1800000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":230400},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":450000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"meta-llama/llama-3.1-8b-instruct":{"id":"meta-llama/llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.08,"cache_read":0.025}},"meta-llama/llama-guard-4-12b":{"id":"meta-llama/llama-guard-4-12b","name":"Llama Guard 4 12B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-30","last_updated":"2025-04-30","modalities":{"input":["image","text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":16384},"cost":{"input":0.18,"output":0.18}},"meta-llama/llama-3.2-3b-instruct":{"id":"meta-llama/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.05,"output":0.33}},"meta-llama/llama-3.2-1b-instruct":{"id":"meta-llama/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60000,"output":54000},"cost":{"input":0.027,"output":0.201}},"meta-llama/llama-4-maverick":{"id":"meta-llama/llama-4-maverick","name":"Llama 4 Maverick","description":"Open multimodal Llama model for strong reasoning and fast responses","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":115200},"cost":{"input":0.2,"output":0.696}},"meta-llama/llama-4-scout":{"id":"meta-llama/llama-4-scout","name":"Llama 4 Scout","description":"Open multimodal Llama model for long-context analysis and efficient agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":16384},"cost":{"input":0.1,"output":0.3}},"meta-llama/llama-3.1-70b-instruct":{"id":"meta-llama/llama-3.1-70b-instruct","name":"Llama-3.1-70B-Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.4,"output":0.4}},"meta-llama/llama-3.3-70b-instruct":{"id":"meta-llama/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.32}},"nousresearch/hermes-3-llama-3.1-70b":{"id":"nousresearch/hermes-3-llama-3.1-70b","name":"Hermes 3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-18","last_updated":"2024-08-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.7,"output":0.7}},"nousresearch/hermes-3-llama-3.1-405b":{"id":"nousresearch/hermes-3-llama-3.1-405b","name":"Hermes 3 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"nousresearch","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-08-16","last_updated":"2024-08-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":1,"output":1}},"nousresearch/hermes-4-405b":{"id":"nousresearch/hermes-4-405b","name":"Hermes 4 405B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"hermes","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":1,"output":3}},"openai/o4-mini-high":{"id":"openai/o4-mini-high","name":"o4 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-06-30","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-4o-2024-05-13":{"id":"openai/gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":5,"output":15}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":15,"output":120}},"openai/gpt-4o-mini-2024-07-18":{"id":"openai/gpt-4o-mini-2024-07-18","name":"GPT-4o-mini (2024-07-18)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"o-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-10-31","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o3-mini-high":{"id":"openai/o3-mini-high","name":"o3 Mini High","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-10-31","release_date":"2025-02-12","last_updated":"2025-02-12","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-6-astra-pro":{"id":"openai/gpt-6-astra-pro","name":"GPT-6 Astra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-audio-mini":{"id":"openai/gpt-audio-mini","name":"GPT Audio Mini","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"o-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.6,"output":2.4}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1 Codex","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4-turbo-preview":{"id":"openai/gpt-4-turbo-preview","name":"GPT-4 Turbo Preview","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.2-chat":{"id":"openai/gpt-5.2-chat","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-10","last_updated":"2025-12-10","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5.6-luna-pro":{"id":"openai/gpt-5.6-luna-pro","name":"GPT-5.6 Luna Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.03,"output":0.13,"cache_read":0.03}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5-image":{"id":"openai/gpt-5-image","name":"GPT-5 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-10-01","release_date":"2025-10-14","last_updated":"2025-10-14","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":10,"output":10,"cache_read":1.25}},"openai/gpt-5.6-sol-pro":{"id":"openai/gpt-5.6-sol-pro","name":"GPT-5.6 Sol Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":15,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":15,"cache_read":0.4,"cache_write":5}}},"openai/gpt-oss-safeguard-20b":{"id":"openai/gpt-oss-safeguard-20b","name":"gpt-oss-safeguard-20b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/gpt-5.4-image-2":{"id":"openai/gpt-5.4-image-2","name":"GPT-5.4 Image 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["image","text","pdf"],"output":["image","text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":8,"output":15,"cache_read":2}},"openai/gpt-3.5-turbo-0613":{"id":"openai/gpt-3.5-turbo-0613","name":"GPT-3.5 Turbo (older v0613)","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1,"output":2}},"openai/gpt-audio":{"id":"openai/gpt-audio","name":"GPT Audio","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text","audio"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.6-terra-pro":{"id":"openai/gpt-5.6-terra-pro","name":"GPT-5.6 Terra Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-chat-latest":{"id":"openai/gpt-chat-latest","name":"GPT Chat Latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-05-05","last_updated":"2026-05-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo-16k":{"id":"openai/gpt-3.5-turbo-16k","name":"GPT-3.5 Turbo 16k","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2023-08-28","last_updated":"2023-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":3,"output":4}},"openai/gpt-5-image-mini":{"id":"openai/gpt-5-image-mini","name":"GPT-5 Image Mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["pdf","image","text"],"output":["image","text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":2.5,"output":2,"cache_read":0.25}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.037,"output":0.17}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-3.5-turbo-instruct":{"id":"openai/gpt-3.5-turbo-instruct","name":"GPT-3.5 Turbo Instruct","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2021-09-30","release_date":"2023-09-28","last_updated":"2023-09-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":4095,"output":3685},"cost":{"input":1.5,"output":2}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":4096},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"~z-ai/glm-flash-latest":{"id":"~z-ai/glm-flash-latest","name":"GLM Flash Latest","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015}},"~z-ai/glm-latest":{"id":"~z-ai/glm-latest","name":"GLM Latest","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1310720,"output":128000},"cost":{"input":0.7,"output":2.2,"cache_read":0.13}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-12-31","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":100352},"cost":{"input":0.6,"output":2.5}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.71,"output":3.5,"cache_read":0.15}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":100352},"cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":943718},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2":{"id":"moonshotai/kimi-k2","name":"Kimi K2 0711","description":"Kimi model for long-context chat, coding, and agentic reasoning","family":"kimi-k2","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-12-31","release_date":"2025-07-11","last_updated":"2025-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":100352},"cost":{"input":0.57,"output":2.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"cohere/north-mini-code:free":{"id":"cohere/north-mini-code:free","name":"North Mini Code (free)","description":"Cohere coding model for practical software engineering and agentic edits","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-17","last_updated":"2026-06-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a":{"id":"cohere/command-a","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8192},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"upstage/solar-pro-3":{"id":"upstage/solar-pro-3","name":"Solar Pro 3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":117964},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"upstage/solar-pro4":{"id":"upstage/solar-pro4","name":"Solar Pro 4","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.03,"output":0.12,"cache_read":0.006}},"arcee-ai/trinity-large-thinking":{"id":"arcee-ai/trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":80000},"cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"tencent/hy3":{"id":"tencent/hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"input":192000,"output":128000},"cost":{"input":0.0825,"output":0.33,"cache_read":0.020625}},"tencent/hy4-preview":{"id":"tencent/hy4-preview","name":"Hy4 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"tencent/hy-mt2-30b-a3b":{"id":"tencent/hy-mt2-30b-a3b","name":"Hy-MT2-30B-A3B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-7b":{"id":"tencent/hy-mt2-7b","name":"Hy-MT2-7B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-08-19","last_updated":"2026-08-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.074,"output":0.295}},"tencent/hy-mt2-1.8b":{"id":"tencent/hy-mt2-1.8b","name":"Hy-MT2-1.8B","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-08-20","last_updated":"2026-08-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":4096},"cost":{"input":0.044,"output":0.177}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":235929},"cost":{"input":0.18,"output":0.6,"cache_read":0.06}},"tencent/hunyuan-a13b-instruct":{"id":"tencent/hunyuan-a13b-instruct","name":"Hunyuan A13B Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-07-08","last_updated":"2025-07-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":117964},"cost":{"input":0.14,"output":0.57}},"liquid/lfm-2.5-2.6b:free":{"id":"liquid/lfm-2.5-2.6b:free","name":"LFM2.5-2.6B (free)","description":"Free provider route for experiments, demos, and cost-sensitive chat workloads","family":"liquid","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":8192},"cost":{"input":0,"output":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.13,"output":0.85,"cache_read":0.025}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":16384},"cost":{"input":0.43,"output":1.75,"cache_read":0.08}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.055}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.28,"output":0.88,"cache_read":0.052}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"z-ai/glm-4.5v":{"id":"z-ai/glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":0.6,"output":1.8,"cache_read":0.11}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.6,"output":1.92,"cache_read":0.12}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":128000},"cost":{"input":0.966,"output":3.036,"cache_read":0.1794}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":false,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":943718},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-4.7-flash":{"id":"z-ai/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_details"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":117964},"cost":{"input":0.0605,"output":0.4}},"cognitivecomputations/dolphin-mistral-24b-venice-edition":{"id":"cognitivecomputations/dolphin-mistral-24b-venice-edition","name":"Uncensored","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04-30","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.2,"output":0.9}},"perplexity/sonar-pro-search":{"id":"perplexity/sonar-pro-search","name":"Sonar Pro Search","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-10-30","last_updated":"2025-10-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar":{"id":"perplexity/sonar","name":"Sonar","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":114364},"cost":{"input":1,"output":1}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded reasoning model for multi-step research and cited answers","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Advanced Sonar search model for deeper research and cited synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8000},"cost":{"input":3,"output":15}},"perplexity/sonar-deep-research":{"id":"perplexity/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","family":"sonar-deep-research","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-03-07","last_updated":"2025-03-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":115200},"cost":{"input":2,"output":8,"reasoning":3}},"rekaai/reka-edge":{"id":"rekaai/reka-edge","name":"Reka Edge","description":"Multimodal model for analyzing text, images, documents, and rich media","family":"reka","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":14745},"cost":{"input":0.1,"output":0.1}},"rekaai/reka-flash-3":{"id":"rekaai/reka-flash-3","name":"Reka Flash 3","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"reka","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01-31","release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":58982},"cost":{"input":0.1,"output":0.2}}}},"cline-pass":{"id":"cline-pass","env":["CLINE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cline.bot/api/v1","name":"ClinePass","doc":"https://docs.cline.bot/getting-started/clinepass","models":{"cline-pass/qwen3.7-max":{"id":"cline-pass/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"cline-pass/kimi-k2.6":{"id":"cline-pass/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"cline-pass/glm-5.2":{"id":"cline-pass/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"cline-pass/minimax-m3":{"id":"cline-pass/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"cline-pass/deepseek-v4-flash":{"id":"cline-pass/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"cline-pass/kimi-k2.7-code":{"id":"cline-pass/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"cline-pass/kimi-k3":{"id":"cline-pass/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"cline-pass/glm-5.3-flash":{"id":"cline-pass/glm-5.3-flash","name":"cline-pass/glm-5.3-flash","description":"Latest natively multimodal model in the GLM-5 series","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"cline-pass/qwen3.8-max":{"id":"cline-pass/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"cline-pass/qwen3.7-plus":{"id":"cline-pass/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5}}},"cline-pass/deepseek-v4-pro":{"id":"cline-pass/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.0145}},"cline-pass/glm-5.3":{"id":"cline-pass/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"cline-pass/mimo-v2.5":{"id":"cline-pass/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"cline-pass/mimo-v2.5-pro":{"id":"cline-pass/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.74,"output":3.48,"cache_read":0.0145}}}},"iteracompute":{"id":"iteracompute","env":["ITERACOMPUTE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.iteracompute.com/v1","name":"IteraCompute","doc":"https://iteracompute.com/docs.html","models":{"iteracompute/qwen3.8-27b":{"id":"iteracompute/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"input":262144,"output":65536},"cost":{"input":0.3,"output":2.5}},"iteracompute/ornith-1.5-35b-a3b":{"id":"iteracompute/ornith-1.5-35b-a3b","name":"Ornith 1.5 35B A3B","description":"Mixture-of-experts coding-reasoning model for agentic software tasks, tool use, and image understanding","family":"ornith","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"input":262144,"output":65536},"cost":{"input":0.3,"output":3}}}},"model-oracle-ai":{"id":"model-oracle-ai","env":["MODEL_ORACLE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.modeloracle.com/api/v1","name":"Model Oracle AI","doc":"https://modeloracle.com/setup/","models":{"claude-opus-4.8":{"id":"claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}},"claude-haiku-4.5":{"id":"claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"auto":{"id":"auto","name":"Auto","description":"Model Oracle AI decision engine that selects and routes among configured coding-agent models","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-29","last_updated":"2026-07-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000}}}},"ofox":{"id":"ofox","env":["OFOX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ofox.ai/v1","name":"Ofox","doc":"https://ofox.ai/docs","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1064000,"output":64000},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.05,"cache_read":0.29}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1131072,"output":131072},"cost":{"input":0.45,"output":3.2,"cache_read":0.05,"cache_write":0.5625}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":1.83,"cache_read":0.29}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.55,"output":3.5,"cache_read":0.55}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.6-flash":{"id":"qwen/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.31}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.6-max-preview":{"id":"qwen/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":2.15,"output":12.86,"cache_read":0.2,"cache_write":1.17}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1064000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.4}},"bailian/qwen3.7-max":{"id":"bailian/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"bailian/qwen3-coder-plus":{"id":"bailian/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.8,"output":9,"cache_read":0.2,"cache_write":1}},"bailian/qwen-vl-max":{"id":"bailian/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.23,"output":0.58,"cache_read":0.046}},"bailian/qwen3-coder-flash":{"id":"bailian/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":2.5,"cache_read":0.06,"cache_write":0.27}},"bailian/qwen-max":{"id":"bailian/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":0.35,"output":1.38,"cache_read":0.069}},"bailian/qwen3.6-plus":{"id":"bailian/qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625}},"bailian/qwen3.5-27b":{"id":"bailian/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":2.05,"cache_read":0.29}},"bailian/qwen3.8-27b":{"id":"bailian/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1131072,"output":131072},"cost":{"input":0.45,"output":3.2,"cache_read":0.05,"cache_write":0.5625}},"bailian/qwen3.5-35b-a3b":{"id":"bailian/qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.29,"output":1.83,"cache_read":0.29}},"bailian/qwen-flash":{"id":"bailian/qwen-flash","name":"Qwen Flash","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.022,"output":0.22,"cache_read":0.0043,"cache_write":0.027}},"bailian/qwen-turbo":{"id":"bailian/qwen-turbo","name":"Qwen Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-11-01","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.05,"output":0.09,"cache_read":0.0086}},"bailian/qwen3.5-flash":{"id":"bailian/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.01,"cache_write":0.125}},"bailian/qwen3-coder-next":{"id":"bailian/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.2,"output":1.5}},"bailian/qwen3.5-397b-a17b":{"id":"bailian/qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.55,"output":3.5,"cache_read":0.55}},"bailian/qwen3.6-27b":{"id":"bailian/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.6,"output":3.6}},"bailian/qwen3.8-max-0902":{"id":"bailian/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"bailian/qwen3-max":{"id":"bailian/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.36,"output":1.43,"cache_read":0.072}},"bailian/qwen-plus":{"id":"bailian/qwen-plus","name":"Qwen Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-09-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":32768},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"bailian/qwen3.5-122b-a10b":{"id":"bailian/qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.29,"output":2.29,"cache_read":0.29}},"bailian/qwen3.6-flash":{"id":"bailian/qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":0.31}},"bailian/qwen3.6-max-preview":{"id":"bailian/qwen3.6-max-preview","name":"Qwen3.6 Max Preview","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":2.15,"output":12.86,"cache_read":0.2,"cache_write":1.17}},"bailian/qwen3.8-max":{"id":"bailian/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"bailian/qwen3.7-plus":{"id":"bailian/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5}},"bailian/qwen3.5-plus":{"id":"bailian/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.4}},"volcengine/doubao-seed-2.1-turbo":{"id":"volcengine/doubao-seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.3536,"output":1.7696,"cache_read":0.068,"cache_write":0.0019}},"volcengine/doubao-seed-2.0-mini":{"id":"volcengine/doubao-seed-2.0-mini","name":"Seed 2.0 Mini","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.06,"output":0.56,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-2.1-pro":{"id":"volcengine/doubao-seed-2.1-pro","name":"Seed 2.1 Pro","description":"Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.7072,"output":3.536,"cache_read":0.1416,"cache_write":0.002}},"volcengine/doubao-seed-2.0-code":{"id":"volcengine/doubao-seed-2.0-code","name":"Seed 2.0 Code","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.67,"output":3.36,"cache_read":0.14,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-pro":{"id":"volcengine/doubao-seed-2.0-pro","name":"Seed 2.0 Pro","description":"Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"cost":{"input":0.67,"output":3.36,"cache_read":0.14,"cache_write":0.0024}},"volcengine/doubao-seed-1-8":{"id":"volcengine/doubao-seed-1-8","name":"Seed 1.8","description":"ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-28","last_updated":"2025-12-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"volcengine/doubao-seed-evolving":{"id":"volcengine/doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.884,"output":4.42,"cache_read":0.177,"cache_write":0.0025}},"volcengine/doubao-seed-character":{"id":"volcengine/doubao-seed-character","name":"Seed Character","description":"ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.177,"output":0.884,"cache_read":0.024,"cache_write":0.0025}},"volcengine/doubao-seed-1-6-vision":{"id":"volcengine/doubao-seed-1-6-vision","name":"Seed 1.6 Vision","description":"ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks","family":"seed","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-15","last_updated":"2025-08-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.12,"output":1.15,"cache_read":0.023}},"volcengine/doubao-seed-1-6-flash":{"id":"volcengine/doubao-seed-1-6-flash","name":"Seed 1.6 Flash","description":"Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use","family":"seed","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.03,"output":0.22,"cache_read":0.0043}},"volcengine/doubao-seed-2.0-lite":{"id":"volcengine/doubao-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.13,"output":0.76,"cache_read":0.03,"cache_write":0.0024}},"volcengine/doubao-seed-1-6":{"id":"volcengine/doubao-seed-1-6","name":"Seed 1.6","description":"ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks","family":"seed","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.12,"output":0.29,"cache_read":0.023}},"minimax/minimax-m2.1-lightning":{"id":"minimax/minimax-m2.1-lightning","name":"MiniMax-M2.1 Lightning","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/m2-her":{"id":"minimax/m2-her","name":"MiniMax-M2 Her","description":"MiniMax M2 variant tuned for conversational and character-driven agent interactions","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":2048},"cost":{"input":0.3,"output":1.2}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"Low-latency M2.7 variant for interactive coding plans and agent loops","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.6,"output":2.4,"cache_read":0.12}},"minimax/minimax-m2.5-lightning":{"id":"minimax/minimax-m2.5-lightning","name":"MiniMax-M2.5 Lightning","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":2.4,"cache_read":0.03,"cache_write":0.375}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-fable-5.1":{"id":"anthropic/claude-fable-5.1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://api.ofox.ai/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":512,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.1,"output":0.4,"cache_read":0.025,"cache_write":1,"input_audio":0.3}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":1.5}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"cache_write":1,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":1.5,"output":9,"cache_read":0.15,"cache_write":0.083,"input_audio":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":0.083}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":1}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"provider":{"npm":"@ai-sdk/google","api":"https://api.ofox.ai/gemini/v1beta"},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.0415,"input_audio":1.5}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.25,"output":10,"cache_read":0.125,"cache_write":4.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":0,"max":24576}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03,"cache_write":1,"input_audio":1}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"status":"beta","cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-flash-0731":{"id":"deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-v4.1-flash":{"id":"deepseek/deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.29,"output":0.43,"cache_read":0.06}},"deepseek/deepseek-v4-pro-0423":{"id":"deepseek/deepseek-v4-pro-0423","name":"DeepSeek V4 Pro 0423","description":"DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2}},"x-ai/grok-4.20":{"id":"x-ai/grok-4.20","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":128000},"cost":{"input":4,"output":12,"cache_read":0.4}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.3}},"x-ai/grok-4.6":{"id":"x-ai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":65536},"cost":{"input":2,"output":6,"cache_read":0.5}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"Grok 4.1 Fast","description":"xAI's fast agentic tool-calling model with a 2M context window; non-reasoning variant for low-latency responses","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.05,"output":0.4,"cache_read":0.01}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":65536},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.18}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.1-codex-max":{"id":"openai/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.18}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32768},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":30,"output":180}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.18}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.13}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://api.ofox.ai/v1"},"cost":{"input":5,"output":30,"cache_read":0.5}},"moonshotai/kimi-k2.7-code-highspeed":{"id":"moonshotai/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3,"cache_read":0.1}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.4,"output":1.9,"cache_read":0.11}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5.3-flash":{"id":"z-ai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"z-ai/glm-4.7-flashx":{"id":"z-ai/glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.072,"output":0.43,"cache_read":0.015}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"z-ai/glm-5.3":{"id":"z-ai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}}}},"arcee":{"id":"arcee","env":["ARCEE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.arcee.ai/api/v1","name":"Arcee","doc":"https://docs.arcee.ai","models":{"trinity-large-thinking":{"id":"trinity-large-thinking","name":"Trinity Large Thinking","description":"Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use","family":"trinity","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0.25,"output":0.8,"cache_read":0.06}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"thinkingmachines/inkling-small":{"id":"thinkingmachines/inkling-small","name":"Inkling Small","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"deepseek/deepseek-v4-pro-0813":{"id":"deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"deepseek/deepseek-v4-flash-latest":{"id":"deepseek/deepseek-v4-flash-latest","name":"DeepSeek V4 Flash Latest","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"status":"beta","cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":384000},"status":"beta","cost":{"input":1.74,"output":3.48,"cache_read":0.2}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"status":"beta","cost":{"input":3,"output":15,"cache_read":0.3}}}},"kuae-cloud-coding-plan":{"id":"kuae-cloud-coding-plan","env":["KUAE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://coding-plan-endpoint.kuaecloud.net/v1","name":"KUAE Cloud Coding Plan","doc":"https://docs.mthreads.com/kuaecloud/kuaecloud-doc-online/coding_plan/","models":{"GLM-4.7":{"id":"GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"ebcloud":{"id":"ebcloud","env":["EBCLOUD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://maas-api.ebcloud.com/v1","name":"EBCloud","doc":"https://docs.ebtech.com/ai/model-api.html","models":{"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.143,"output":0.2857}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.8571,"output":3.4286}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.9286,"output":3.8571}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.4286,"output":0.8571}}}},"agnes":{"id":"agnes","env":["AGNES_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://apihub.agnes-ai.com/v1","name":"Agnes AI","doc":"https://agnes-ai.com/doc","models":{"agnes-2.5-pro-alpha":{"id":"agnes-2.5-pro-alpha","name":"Agnes 2.5 Pro Alpha","description":"Paid reasoning model for advanced coding, scientific reasoning, long-context analysis, agentic workflows, and multimodal understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":0.45,"output":0.9,"cache_read":0.0038}},"agnes-2.5-flash":{"id":"agnes-2.5-flash","name":"Agnes 2.5 Flash","description":"Upgraded model with improved coding, agent workflows, tool calling, and multimodal understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07","last_updated":"2026-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":65536},"cost":{"input":0,"output":0}},"agnes-2.0-flash":{"id":"agnes-2.0-flash","name":"Agnes 2.0 Flash","description":"Fast and efficient model for agent workflows, tool calling, coding, and image understanding.","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-25","last_updated":"2026-05-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":65536},"cost":{"input":0,"output":0}}}},"amd":{"id":"amd","env":["AMD_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://developer.amd.com.cn/radeon/api/v1","name":"AMD","doc":"https://developer.amd.com.cn/radeon/tokenfactory","models":{"Qwen3.8-Flash-Next":{"id":"Qwen3.8-Flash-Next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016}},"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"DeepSeek-V4-Flash-Vision-Exp":{"id":"DeepSeek-V4-Flash-Vision-Exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"MiniCPM5-1B":{"id":"MiniCPM5-1B","name":"MiniCPM5-1B","description":"Dense 1B-class open-source model for on-device and resource-constrained use, with native long-context support, Think / No Think chat modes, and tool calling","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.124,"output":0.7425,"cache_read":0.124}}}},"xiaomi-token-plan-sgp":{"id":"xiaomi-token-plan-sgp","env":["XIAOMI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan-sgp.xiaomimimo.com/v1","name":"Xiaomi Token Plan (Singapore)","doc":"https://platform.xiaomimimo.com/#/docs","models":{"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo-V2-Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"mimo-v2.5-tts-voicedesign":{"id":"mimo-v2.5-tts-voicedesign","name":"MiMo-V2.5-TTS-VoiceDesign","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2-tts":{"id":"mimo-v2-tts","name":"MiMo-V2-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts-voiceclone":{"id":"mimo-v2.5-tts-voiceclone","name":"MiMo-V2.5-TTS-VoiceClone","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}},"mimo-v2.5-tts":{"id":"mimo-v2.5-tts","name":"MiMo-V2.5-TTS","description":"Speech generation model for controllable voice, narration, and audio delivery","family":"mimo","attachment":false,"reasoning":false,"tool_call":false,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["audio"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0,"output":0}}}},"neon":{"id":"neon","env":["NEON_AI_GATEWAY_BASE_URL","NEON_AI_GATEWAY_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"${NEON_AI_GATEWAY_BASE_URL}/v1","name":"Neon","doc":"https://neon.com/docs","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"gpt-5-6-terra":{"id":"gpt-5-6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"qwen35-122b-a10b":{"id":"qwen35-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":25000},"cost":{"input":0.22,"output":2.2}},"qwen3-next-80b-a3b-instruct":{"id":"qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":10000},"cost":{"input":0.15,"output":1.2}},"gpt-5-4-nano":{"id":"gpt-5-4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gemini-3-5-flash":{"id":"gemini-3-5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15,"input_audio":1.5}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-6-sol":{"id":"gpt-5-6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"meta-llama-3-3-70b-instruct":{"id":"meta-llama-3-3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.5,"output":1.5}},"gpt-5-3-codex":{"id":"gpt-5-3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5-6-luna":{"id":"gpt-5-6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-6-flash":{"id":"gemini-3-6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5,"cache_read":0.15,"input_audio":1.5}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":25000},"cost":{"input":0.07,"output":0.3}},"gemini-3-1-pro":{"id":"gemini-3-1-pro","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"cache_read":0.2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"gpt-5-2":{"id":"gpt-5-2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gemini-3-1-flash-lite":{"id":"gemini-3-1-flash-lite","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025,"input_audio":0.5}},"llama-4-maverick":{"id":"llama-4-maverick","name":"Llama 4 Maverick 17B Instruct","description":"Open multimodal Llama for strong reasoning with efficient everyday serving","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":8192},"cost":{"input":0.5,"output":1.5}},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gemini-3-5-flash-lite":{"id":"gemini-3-5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"cache_read":0.03}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":3,"output":15,"cache_read":0.3}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"gpt-5-5":{"id":"gpt-5-5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3 Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"meta-llama-3-1-8b-instruct":{"id":"meta-llama-3-1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Meta's compact open-weight Llama 3.1 model for fast, low-cost text generation","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12-31","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.45}},"gpt-5-4-mini":{"id":"gpt-5-4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"glm-5-2":{"id":"glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":65536},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"inkling":{"id":"inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":65536},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-1":{"id":"claude-opus-4-1","name":"Claude Opus 4.1 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"gpt-5-1":{"id":"gpt-5-1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5-5-pro":{"id":"gpt-5-5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"grok-4-6":{"id":"grok-4-6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":524288},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":25000},"cost":{"input":0.15,"output":0.6}},"gemini-3-flash":{"id":"gemini-3-flash","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"input_audio":1}},"gemma-3-12b":{"id":"gemma-3-12b","name":"Gemma 3 12B","description":"Google's open-weight Gemma 3 vision-language model for text and image understanding","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-08-31","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.15,"output":0.5}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"gpt-5-4":{"id":"gpt-5-4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"${NEON_AI_GATEWAY_BASE_URL}/openai/v1","shape":"responses"},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}}}},"qihang-ai":{"id":"qihang-ai","env":["QIHANG_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.qhaigc.net/v1","name":"QiHang","doc":"https://www.qhaigc.net/docs","models":{"gpt-5.2-codex":{"id":"gpt-5.2-codex","name":"GPT-5.2 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.14,"output":1.14}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-11","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65000},"cost":{"input":0.57,"output":3.43}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.43,"output":2.14}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.14,"output":0.71}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.07,"output":0.43,"tiers":[{"input":0.07,"output":0.43,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.07,"output":0.43}}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5-Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-09-15","last_updated":"2025-09-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.04,"output":0.29}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.09,"output":0.71,"tiers":[{"input":0.09,"output":0.71,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":0.09,"output":0.71}}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":0.71,"output":3.57}}}},"scnet-token-plan":{"id":"scnet-token-plan","env":["SCNET_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.scnet.cn/api/llm/v1","name":"SCNet Token Plan","doc":"https://www.scnet.cn/ac/openapi/doc/2.0/moduleapi/plans/token-plan.html","models":{"DeepSeek-V4-Flash":{"id":"DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.1":{"id":"GLM-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Qwen3.8-Max":{"id":"Qwen3.8-Max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Flash-0731":{"id":"DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"Qwen3.8-Flash":{"id":"Qwen3.8-Flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.5":{"id":"Kimi-K2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.3":{"id":"GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.7-Code":{"id":"Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.2":{"id":"GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5":{"id":"GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K2.6":{"id":"Kimi-K2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Pro-0813":{"id":"DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"Kimi-K3":{"id":"Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"GLM-5.3-Flash":{"id":"GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"DeepSeek-V4-Pro":{"id":"DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"inference":{"id":"inference","env":["INFERENCE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.net/v1","name":"Inference","doc":"https://inference.net/models","models":{"qwen/qwen-2.5-7b-vision-instruct":{"id":"qwen/qwen-2.5-7b-vision-instruct","name":"Qwen 2.5 7B Vision Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":125000,"output":4096},"cost":{"input":0.2,"output":0.2}},"qwen/qwen3-embedding-4b":{"id":"qwen/qwen3-embedding-4b","name":"Qwen 3 Embedding 4B","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":2048},"cost":{"input":0.01,"output":0}},"google/gemma-3":{"id":"google/gemma-3","name":"Google Gemma 3","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":125000,"output":4096},"cost":{"input":0.15,"output":0.3}},"meta/llama-3.1-8b-instruct":{"id":"meta/llama-3.1-8b-instruct","name":"Llama 3.1 8B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.025,"output":0.025}},"meta/llama-3.2-3b-instruct":{"id":"meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.02,"output":0.02}},"meta/llama-3.2-1b-instruct":{"id":"meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.01,"output":0.01}},"meta/llama-3.2-11b-vision-instruct":{"id":"meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.055,"output":0.055}},"osmosis/osmosis-structure-0.6b":{"id":"osmosis/osmosis-structure-0.6b","name":"Osmosis Structure 0.6B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"osmosis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":4000,"output":2048},"cost":{"input":0.1,"output":0.5}},"mistral/mistral-nemo-12b-instruct":{"id":"mistral/mistral-nemo-12b-instruct","name":"Mistral Nemo 12B Instruct","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-01-01","last_updated":"2025-01-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4096},"cost":{"input":0.038,"output":0.1}}}},"openai":{"id":"openai","env":["OPENAI_API_KEY"],"npm":"@ai-sdk/openai","name":"OpenAI","doc":"https://platform.openai.com/docs/models","models":{"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"status":"deprecated","cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"gpt-4o-2024-05-13":{"id":"gpt-4o-2024-05-13","name":"GPT-4o (2024-05-13)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":5,"output":15}},"gpt-5-pro":{"id":"gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"chatgpt-image-latest":{"id":"chatgpt-image-latest","name":"chatgpt-image-latest","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"gpt-4o-2024-08-06":{"id":"gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":20,"output":100,"cache_read":2,"cache_write":25},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"gpt-5.2-pro":{"id":"gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"gpt-5.3-codex-spark":{"id":"gpt-5.3-codex-spark","name":"GPT-5.3 Codex Spark","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex-spark","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"input":100000,"output":32000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"gpt-4-turbo":{"id":"gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"status":"deprecated","cost":{"input":10,"output":30}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"o1":{"id":"o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":15,"output":60,"cache_read":7.5}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":0.4,"output":2.4,"cache_read":0.04,"cache_write":0.5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"gpt-5.3-codex":{"id":"gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"gpt-image-1.5":{"id":"gpt-image-1.5","name":"gpt-image-1.5","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"o1-pro":{"id":"o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":150,"output":600}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"text-embedding-ada-002":{"id":"text-embedding-ada-002","name":"text-embedding-ada-002","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2022-12","release_date":"2022-12-15","last_updated":"2022-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":1536},"cost":{"input":0.1,"output":0}},"gpt-image-1":{"id":"gpt-image-1","name":"gpt-image-1","description":"OpenAI image model for production generation, edits, and brand-safe visual workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-04-24","last_updated":"2025-04-24","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0},"status":"deprecated"},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"gpt-5.5-pro":{"id":"gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"gpt-image-1-mini":{"id":"gpt-image-1-mini","name":"gpt-image-1-mini","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":1.5,"output":9,"cache_read":0.15},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"gpt-image-2":{"id":"gpt-image-2","name":"gpt-image-2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gpt-image","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["image"]},"open_weights":false,"limit":{"context":0,"input":0,"output":0},"cost":{"input":5,"output":30,"cache_read":1.25}},"gpt-3.5-turbo":{"id":"gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"status":"deprecated","cost":{"input":0.5,"output":1.5,"cache_read":0}},"gpt-5.6":{"id":"gpt-5.6","name":"GPT-5.6","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":8,"output":40,"cache_read":0.8,"cache_write":10},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"text-embedding-3-small":{"id":"text-embedding-3-small","name":"text-embedding-3-small","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-01","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":1536},"cost":{"input":0.02,"output":0}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"gpt-5.4-pro":{"id":"gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"text-embedding-3-large":{"id":"text-embedding-3-large","name":"text-embedding-3-large","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"text-embedding","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"knowledge":"2024-01","release_date":"2024-01-25","last_updated":"2024-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":3072},"cost":{"input":0.13,"output":0}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":4,"output":24,"cache_read":0.4,"cache_write":5},"provider":{"body":{"service_tier":"priority"}}},"pro":{"provider":{"body":{"reasoning":{"mode":"pro"}}}}}},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"gpt-4":{"id":"gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8192,"output":8192},"status":"deprecated","cost":{"input":30,"output":60}},"gpt-5.2":{"id":"gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"GPT-5.2 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"o4-mini":{"id":"o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"gpt-realtime-2.1":{"id":"gpt-realtime-2.1","name":"GPT-Realtime-2.1","description":"Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text","audio","image"],"output":["text","audio"]},"open_weights":false,"limit":{"context":128000,"input":96000,"output":32000},"cost":{"input":4,"output":24,"cache_read":0.4,"input_audio":32,"output_audio":64}},"o3-mini":{"id":"o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"status":"deprecated","cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"o3-pro":{"id":"o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"status":"deprecated","cost":{"input":1.75,"output":14,"cache_read":0.175}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"gpt-4o-2024-11-20":{"id":"gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}}}},"aiand":{"id":"aiand","env":["AIAND_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.aiand.com/v1","name":"ai&","doc":"https://docs.aiand.com/","models":{"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":3,"cache_read":0.2}},"qwen/qwen3.6-27b":{"id":"qwen/qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.32,"output":3.2}},"motif-technologies/motif-3":{"id":"motif-technologies/motif-3","name":"Motif 3","description":"Motif 3 is a large-scale, decoder-only Mixture-of-Experts (MoE) language model with 314 billion total parameters and 13.2 billion parameters activated per token.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2}},"deepseek-ai/deepseek-v4-flash":{"id":"deepseek-ai/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.15,"output":0.25}},"deepseek-ai/deepseek-v4-pro":{"id":"deepseek-ai/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1,"output":2.5}},"google/gemma-4-31b-it":{"id":"google/gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.2,"output":0.5}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":4}},"zai-org/glm-5.3":{"id":"zai-org/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":4,"cache_read":0.3}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":12.5,"cache_read":0.5}}}},"siliconflow":{"id":"siliconflow","env":["SILICONFLOW_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.siliconflow.com/v1","name":"SiliconFlow","doc":"https://cloud.siliconflow.com/models","models":{"baidu/ERNIE-4.5-300B-A47B":{"id":"baidu/ERNIE-4.5-300B-A47B","name":"baidu/ERNIE-4.5-300B-A47B","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"ernie","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-02","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.28,"output":1.1}},"stepfun-ai/Step-3.5-Flash":{"id":"stepfun-ai/Step-3.5-Flash","name":"stepfun-ai/Step-3.5-Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","family":"step","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.1,"output":0.3}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"deepseek-ai/DeepSeek-V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-12-26","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.25,"output":1}},"deepseek-ai/DeepSeek-V4-Flash":{"id":"deepseek-ai/DeepSeek-V4-Flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"deepseek-ai/DeepSeek-V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3.1-Terminus":{"id":"deepseek-ai/DeepSeek-V3.1-Terminus","name":"deepseek-ai/DeepSeek-V3.1-Terminus","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-29","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-V3.2-Exp":{"id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"deepseek-ai/DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.41}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"deepseek-ai/DeepSeek-R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-05-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.5,"output":2.18}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"deepseek-ai/DeepSeek-V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.42}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"inclusionAI/Ling-flash-2.0":{"id":"inclusionAI/Ling-flash-2.0","name":"inclusionAI/Ling-flash-2.0","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.13,"output":0.4}},"google/gemma-4-26B-A4B-it":{"id":"google/gemma-4-26B-A4B-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.12,"output":0.4}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"zai-org/GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-5V-Turbo":{"id":"zai-org/GLM-5V-Turbo","name":"zai-org/GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"zai-org/GLM-4.5-Air":{"id":"zai-org/GLM-4.5-Air","name":"zai-org/GLM-4.5-Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-air","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.86}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1049000,"output":262000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"zai-org/GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":205000,"output":205000},"cost":{"input":0.95,"output":2.55,"cache_read":0.2}},"Qwen/Qwen3-30B-A3B-Instruct-2507":{"id":"Qwen/Qwen3-30B-A3B-Instruct-2507","name":"Qwen/Qwen3-30B-A3B-Instruct-2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.3}},"Qwen/Qwen3-VL-30B-A3B-Thinking":{"id":"Qwen/Qwen3-VL-30B-A3B-Thinking","name":"Qwen/Qwen3-VL-30B-A3B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-11","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"Qwen/Qwen3.5-27B":{"id":"Qwen/Qwen3.5-27B","name":"Qwen3.5 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":2}},"Qwen/Qwen3-8B":{"id":"Qwen/Qwen3-8B","name":"Qwen/Qwen3-8B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.06,"output":0.06}},"Qwen/Qwen3-VL-32B-Thinking":{"id":"Qwen/Qwen3-VL-32B-Thinking","name":"Qwen/Qwen3-VL-32B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":1.5}},"Qwen/Qwen3-14B":{"id":"Qwen/Qwen3-14B","name":"Qwen/Qwen3-14B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3-Coder-30B-A3B-Instruct":{"id":"Qwen/Qwen3-Coder-30B-A3B-Instruct","name":"Qwen/Qwen3-Coder-30B-A3B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-01","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.07,"output":0.28}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen/Qwen3.5-9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.15}},"Qwen/Qwen3.5-122B-A10B":{"id":"Qwen/Qwen3.5-122B-A10B","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.26,"output":2.08}},"Qwen/Qwen3-VL-235B-A22B-Instruct":{"id":"Qwen/Qwen3-VL-235B-A22B-Instruct","name":"Qwen/Qwen3-VL-235B-A22B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-04","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.3,"output":1.5}},"Qwen/Qwen3-Coder-480B-A35B-Instruct":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct","name":"Qwen/Qwen3-Coder-480B-A35B-Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-31","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.25,"output":1}},"Qwen/Qwen2.5-7B-Instruct":{"id":"Qwen/Qwen2.5-7B-Instruct","name":"Qwen/Qwen2.5-7B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.05,"output":0.05}},"Qwen/Qwen2.5-72B-Instruct":{"id":"Qwen/Qwen2.5-72B-Instruct","name":"Qwen/Qwen2.5-72B-Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-18","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":33000,"output":4000},"cost":{"input":0.59,"output":0.59}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.39,"output":2.34}},"Qwen/Qwen3.5-35B-A3B":{"id":"Qwen/Qwen3.5-35B-A3B","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.24,"output":1.8}},"Qwen/Qwen3-VL-32B-Instruct":{"id":"Qwen/Qwen3-VL-32B-Instruct","name":"Qwen/Qwen3-VL-32B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-21","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3-32B":{"id":"Qwen/Qwen3-32B","name":"Qwen/Qwen3-32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-04-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"Qwen/Qwen3-235B-A22B-Thinking-2507":{"id":"Qwen/Qwen3-235B-A22B-Thinking-2507","name":"Qwen/Qwen3-235B-A22B-Thinking-2507","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-28","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.13,"output":0.6}},"Qwen/Qwen3.6-27B":{"id":"Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":3.2}},"Qwen/Qwen3-VL-8B-Instruct":{"id":"Qwen/Qwen3-VL-8B-Instruct","name":"Qwen/Qwen3-VL-8B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-15","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.18,"output":0.68}},"Qwen/Qwen3.6-35B-A3B":{"id":"Qwen/Qwen3.6-35B-A3B","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.2,"output":1.6}},"Qwen/Qwen3-VL-235B-A22B-Thinking":{"id":"Qwen/Qwen3-VL-235B-A22B-Thinking","name":"Qwen/Qwen3-VL-235B-A22B-Thinking","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-04","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":3.5}},"Qwen/Qwen3-VL-30B-A3B-Instruct":{"id":"Qwen/Qwen3-VL-30B-A3B-Instruct","name":"Qwen/Qwen3-VL-30B-A3B-Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-05","last_updated":"2025-11-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.29,"output":1}},"ByteDance-Seed/Seed-OSS-36B-Instruct":{"id":"ByteDance-Seed/Seed-OSS-36B-Instruct","name":"ByteDance-Seed/Seed-OSS-36B-Instruct","description":"Tool-capable chat model for instruction following and agentic application workflows","family":"seed","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-09-04","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.21,"output":0.57}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMaxAI/MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-06-15","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":197000,"output":131000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"openai/gpt-oss-20b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":8000},"cost":{"input":0.04,"output":0.18}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"openai/gpt-oss-120b","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-13","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":8000},"cost":{"input":0.05,"output":0.45}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"moonshotai/Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"moonshotai/Kimi-K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-21","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.77,"output":4,"cache_read":0.2}},"tencent/Hunyuan-A13B-Instruct":{"id":"tencent/Hunyuan-A13B-Instruct","name":"tencent/Hunyuan-A13B-Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-06-30","last_updated":"2025-11-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.14,"output":0.57}},"tencent/Hy3-preview":{"id":"tencent/Hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":128,"max":32768}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.066,"output":0.26,"cache_read":0.029}}}},"stepfun-ai-step-plan":{"id":"stepfun-ai-step-plan","env":["STEPFUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.stepfun.ai/step_plan/v1","name":"StepFun Step Plan (Global)","doc":"https://platform.stepfun.ai/docs/en/step-plan/integrations/reasoning-api","models":{"step-3.7-flash":{"id":"step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash":{"id":"step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}},"step-3.5-flash-2603":{"id":"step-3.5-flash-2603","name":"Step 3.5 Flash 2603","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000}}}},"hetzner":{"id":"hetzner","env":["HETZNER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.hetzner.com/api/v1","name":"Hetzner","doc":"https://experiments.hetzner.com/docs/inference","models":{"Qwen3.8-27B":{"id":"Qwen3.8-27B","name":"Qwen3.8-27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0,"output":0,"cache_read":0}},"Qwen/Qwen3.6-35B-A3B-FP8":{"id":"Qwen/Qwen3.6-35B-A3B-FP8","name":"Qwen3.6 35B A3B FP8","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"beta","cost":{"input":0,"output":0,"cache_read":0}}}},"snowflake-cortex":{"id":"snowflake-cortex","env":["SNOWFLAKE_ACCOUNT","SNOWFLAKE_CORTEX_PAT"],"npm":"@ai-sdk/openai-compatible","api":"https://${SNOWFLAKE_ACCOUNT}.snowflakecomputing.com/api/v2/cortex/v1","name":"Snowflake Cortex","doc":"https://docs.snowflake.com/en/user-guide/snowflake-cortex/cortex-rest-api","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":16384}},"gemini-3.1-pro":{"id":"gemini-3.1-pro","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"openai-gpt-5.1":{"id":"openai-gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai-gpt-5.6-terra":{"id":"openai-gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-opus-4-5":{"id":"claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"openai-gpt-4.1":{"id":"openai-gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"mistral-large2":{"id":"mistral-large2","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"openai-gpt-5.2":{"id":"openai-gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai-gpt-5.6-luna":{"id":"openai-gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"openai-gpt-5":{"id":"openai-gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"status":"beta"},"claude-opus-4-6":{"id":"claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"deepseek-r1":{"id":"deepseek-r1","name":"DeepSeek-R1","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"status":"beta","experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"openai-gpt-5.6-sol":{"id":"openai-gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"openai-gpt-5.4":{"id":"openai-gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta","experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}}},"openai-gpt-5-nano":{"id":"openai-gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"status":"beta"},"openai-gpt-5.5":{"id":"openai-gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"status":"beta"},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384}},"claude-sonnet-4-5":{"id":"claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":16384}},"openai-gpt-5-mini":{"id":"openai-gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"input":272000,"output":8192},"status":"beta"},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000}},"snowflake-llama3.3-70b":{"id":"snowflake-llama3.3-70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096}}}},"meganova":{"id":"meganova","env":["MEGANOVA_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.meganova.ai/v1","name":"Meganova","doc":"https://docs.meganova.ai","models":{"deepseek-ai/DeepSeek-V3-0324":{"id":"deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.25,"output":0.88}},"deepseek-ai/DeepSeek-V3.1":{"id":"deepseek-ai/DeepSeek-V3.1","name":"DeepSeek V3.1","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-25","last_updated":"2025-08-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":1}},"deepseek-ai/DeepSeek-R1-0528":{"id":"deepseek-ai/DeepSeek-R1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":64000},"cost":{"input":0.5,"output":2.15}},"deepseek-ai/DeepSeek-V3.2-Exp":{"id":"deepseek-ai/DeepSeek-V3.2-Exp","name":"DeepSeek V3.2 Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-10","last_updated":"2025-10-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.27,"output":0.4}},"deepseek-ai/DeepSeek-V3.2":{"id":"deepseek-ai/DeepSeek-V3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.26,"output":0.38}},"mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0,"output":0}},"mistralai/Mistral-Nemo-Instruct-2407":{"id":"mistralai/Mistral-Nemo-Instruct-2407","name":"Mistral Nemo Instruct 2407","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0.02,"output":0.04}},"zai-org/GLM-4.7":{"id":"zai-org/GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.2,"output":0.8}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.8,"output":2.56}},"zai-org/GLM-4.6":{"id":"zai-org/GLM-4.6","name":"GLM-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.45,"output":1.9}},"Qwen/Qwen2.5-VL-32B-Instruct":{"id":"Qwen/Qwen2.5-VL-32B-Instruct","name":"Qwen2.5 VL 32B Instruct","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16384,"output":16384},"cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.5-Plus":{"id":"Qwen/Qwen3.5-Plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02","last_updated":"2026-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":2.4,"reasoning":2.4}},"Qwen/Qwen3-235B-A22B-Instruct-2507":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507","name":"Qwen3 235B A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.09,"output":0.6}},"MiniMaxAI/MiniMax-M2.1":{"id":"MiniMaxAI/MiniMax-M2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":131072},"cost":{"input":0.28,"output":1.2}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"meta-llama/Llama-3.3-70B-Instruct":{"id":"meta-llama/Llama-3.3-70B-Instruct","name":"Llama 3.3 70B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.1,"output":0.3}},"moonshotai/Kimi-K2-Thinking":{"id":"moonshotai/Kimi-K2-Thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":2.6}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2026-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":2.8}},"XiaomiMiMo/MiMo-V2-Flash":{"id":"XiaomiMiMo/MiMo-V2-Flash","name":"MiMo V2 Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32000},"cost":{"input":0.1,"output":0.3}}}},"moonshotai":{"id":"moonshotai","env":["MOONSHOT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.moonshot.ai/v1","name":"Moonshot AI","doc":"https://platform.moonshot.ai/docs/api/chat","models":{"kimi-k2.7-code-highspeed":{"id":"kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code HighSpeed","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"volcengine-coding-plan":{"id":"volcengine-coding-plan","env":["ARK_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ark.cn-beijing.volces.com/api/coding/v3","name":"Volcengine Ark Coding Plan","doc":"https://www.volcengine.com/docs/82379/1928261","models":{"doubao-seed-2.1-turbo":{"id":"doubao-seed-2.1-turbo","name":"Seed 2.1 Turbo","description":"Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0,"cache_read":0}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"doubao-seed-evolving":{"id":"doubao-seed-evolving","name":"Seed Evolving","description":"Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-23","last_updated":"2026-06-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0}},"doubao-seed-2.0-lite":{"id":"doubao-seed-2.0-lite","name":"Seed 2.0 Lite","description":"Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0,"output":0,"cache_read":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}}}},"302ai":{"id":"302ai","env":["302AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.302.ai/v1","name":"302.AI","doc":"https://doc.302.ai","models":{"claude-sonnet-4-6":{"id":"claude-sonnet-4-6","name":"claude-sonnet-4-6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-18","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"ministral-14b-2512":{"id":"ministral-14b-2512","name":"ministral-14b-2512","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.33,"output":0.33}},"glm-4.7":{"id":"glm-4.7","name":"glm-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.286,"output":1.142}},"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.8,"output":5.3}},"gemini-3.5-flash-thinking":{"id":"gemini-3.5-flash-thinking","name":"gemini-3.5-flash-thinking","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"gpt-4.1-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4}},"claude-sonnet-4-6-thinking":{"id":"claude-sonnet-4-6-thinking","name":"claude-sonnet-4-6-thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2026-02-18","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15}},"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-10-26","last_updated":"2025-10-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.33,"output":1.32}},"glm-4.6":{"id":"glm-4.6","name":"glm-4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.286,"output":1.142}},"gpt-5-pro":{"id":"gpt-5-pro","name":"gpt-5-pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-08","last_updated":"2025-10-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"gemini-2.5-flash-image":{"id":"gemini-2.5-flash-image","name":"gemini-2.5-flash-image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2025-10-08","last_updated":"2025-10-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":30}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.145,"output":0.43}},"deepseek-v3.2-thinking":{"id":"deepseek-v3.2-thinking","name":"DeepSeek-V3.2-Thinking","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.29,"output":0.43}},"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.8}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"qwen3.5-35b-a3b":{"id":"qwen3.5-35b-a3b","name":"Qwen3.5 35B-A3B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.06,"output":0.46}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"gemini-3.1-pro-preview":{"id":"gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}},"gpt-6-astra":{"id":"gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50}},"gpt-5.6-luna-pro":{"id":"gpt-5.6-luna-pro","name":"gpt-5.6-luna-pro","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4}},"kimi-k2-thinking":{"id":"kimi-k2-thinking","name":"kimi-k2-thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.575,"output":2.3}},"claude-sonnet-4-5-20250929-thinking":{"id":"claude-sonnet-4-5-20250929-thinking","name":"claude-sonnet-4-5-20250929-thinking","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"gemini-3-pro-preview":{"id":"gemini-3-pro-preview","name":"gemini-3-pro-preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-19","last_updated":"2025-11-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":2,"output":12}},"claude-opus-4-1-20250805":{"id":"claude-opus-4-1-20250805","name":"claude-opus-4-1-20250805","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":31999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6}},"qwen3.7-max-2026-06-08":{"id":"qwen3.7-max-2026-06-08","name":"qwen3.7-max-2026-06-08","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.8,"output":5.3}},"qwen3-max-2025-09-23":{"id":"qwen3-max-2025-09-23","name":"qwen3-max-2025-09-23","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-24","last_updated":"2025-09-24","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":258048,"output":65536},"cost":{"input":0.86,"output":3.43}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"gpt-4.1-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6}},"gemini-2.0-flash-lite":{"id":"gemini-2.0-flash-lite","name":"gemini-2.0-flash-lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-11","release_date":"2025-06-16","last_updated":"2025-06-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":8192},"cost":{"input":0.075,"output":0.3}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":7.5}},"gpt-5.4":{"id":"gpt-5.4","name":"gpt-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":0,"tiers":[{"input":5,"output":22.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5}}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"gpt-5.6-sol-pro":{"id":"gpt-5.6-sol-pro","name":"gpt-5.6-sol-pro","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"grok-4-1-fast-reasoning":{"id":"grok-4-1-fast-reasoning","name":"grok-4-1-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"claude-fable-5-1":{"id":"claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-2.5-flash-preview-09-2025":{"id":"gemini-2.5-flash-preview-09-2025","name":"gemini-2.5-flash-preview-09-2025","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"claude-opus-4-7-thinking":{"id":"claude-opus-4-7-thinking","name":"claude-opus-4-7-thinking","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gpt-5.1":{"id":"gpt-5.1","name":"gpt-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"gpt-5.1-chat-latest":{"id":"gpt-5.1-chat-latest","name":"gpt-5.1-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-14","last_updated":"2025-11-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash-lite-preview-09-2025":{"id":"gemini-2.5-flash-lite-preview-09-2025","name":"gemini-2.5-flash-lite-preview-09-2025","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-09-26","last_updated":"2025-09-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.1,"output":0.4}},"mistral-large-2512":{"id":"mistral-large-2512","name":"mistral-large-2512","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-16","last_updated":"2025-12-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":262144},"cost":{"input":1.1,"output":3.3}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9}},"gpt-4o":{"id":"gpt-4o","name":"gpt-4o","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-05-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10}},"gemini-3.1-flash-lite-preview":{"id":"gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2}},"grok-4-fast-reasoning":{"id":"grok-4-fast-reasoning","name":"grok-4-fast-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"claude-sonnet-4-5-20250929":{"id":"claude-sonnet-4-5-20250929","name":"claude-sonnet-4-5-20250929","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15}},"claude-opus-4-7":{"id":"claude-opus-4-7","name":"claude-opus-4-7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"deepseek-v3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.29,"output":0.43}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","video","audio"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.283,"output":1.705}},"claude-haiku-4-5-20251001":{"id":"claude-haiku-4-5-20251001","name":"claude-haiku-4-5-20251001","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-29","last_updated":"2025-07-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.286,"output":1.142}},"kimi-k2-0905-preview":{"id":"kimi-k2-0905-preview","name":"kimi-k2-0905-preview","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-05","last_updated":"2025-09-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.632,"output":2.53}},"claude-fable-5":{"id":"claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5}},"doubao-seed-1-8-251215":{"id":"doubao-seed-1-8-251215","name":"doubao-seed-1-8-251215","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":224000,"output":64000},"cost":{"input":0.114,"output":0.286}},"grok-4.1":{"id":"grok-4.1","name":"grok-4.1","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2,"output":10}},"gpt-4.1":{"id":"gpt-4.1","name":"gpt-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"gpt-5.4-nano":{"id":"gpt-5.4-nano","name":"gpt-5.4-nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25}},"gpt-5.6-terra-pro":{"id":"gpt-5.6-terra-pro","name":"gpt-5.6-terra-pro","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12}},"gemini-3-pro-image-preview":{"id":"gemini-3-pro-image-preview","name":"gemini-3-pro-image-preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-06","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":64000},"cost":{"input":2,"output":120}},"MiniMax-M1":{"id":"MiniMax-M1","name":"MiniMax-M1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-06-16","last_updated":"2025-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0.132,"output":1.254}},"gemini-2.5-flash-nothink":{"id":"gemini-2.5-flash-nothink","name":"gemini-2.5-flash-nothink","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-24","last_updated":"2025-06-24","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.29,"output":0.86}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.188,"output":1.133}},"qwen3-30b-a3b":{"id":"qwen3-30b-a3b","name":"Qwen3-30B-A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.11,"output":1.08}},"gpt-5-thinking":{"id":"gpt-5-thinking","name":"gpt-5-thinking","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"cost":{"input":1.25,"output":10}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.18,"output":0.564}},"gpt-5.4-mini":{"id":"gpt-5.4-mini","name":"gpt-5.4-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-19","last_updated":"2026-03-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"claude-haiku-4-5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-16","last_updated":"2025-10-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5}},"glm-5":{"id":"glm-5","name":"glm-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.6}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2.16,"output":6.36}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.66,"output":3.3}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4}},"qwen3-235b-a22b":{"id":"qwen3-235b-a22b","name":"Qwen3-235B-A22B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04-29","last_updated":"2025-04-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.29,"output":2.86}},"gemini-3-flash-preview":{"id":"gemini-3-flash-preview","name":"gemini-3-flash-preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.285,"output":1.15}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"qwen3-235b-a22b-instruct-2507","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":65536},"cost":{"input":0.29,"output":1.143}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.72,"output":2.88}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75}},"claude-opus-4-8":{"id":"claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"glm-5-turbo":{"id":"glm-5-turbo","name":"glm-5-turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.72,"output":3.2}},"gpt-5-mini":{"id":"gpt-5-mini","name":"gpt-5-mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2}},"qwen3-coder-480b-a35b-instruct":{"id":"qwen3-coder-480b-a35b-instruct","name":"qwen3-coder-480b-a35b-instruct","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":0.86,"output":3.43}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"gemini-2.5-pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1.25,"output":10}},"claude-opus-5-thinking":{"id":"claude-opus-5-thinking","name":"claude-opus-5-thinking","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25}},"gemini-3.1-flash-image-preview":{"id":"gemini-3.1-flash-image-preview","name":"gemini-3.1-flash-image-preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-27","last_updated":"2026-02-27","modalities":{"input":["text","image","pdf"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":60}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.72,"output":3.2}},"gpt-5.2":{"id":"gpt-5.2","name":"gpt-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14}},"doubao-seed-1-6-vision-250815":{"id":"doubao-seed-1-6-vision-250815","name":"doubao-seed-1-6-vision-250815","description":"Multimodal model for analyzing text, images, documents, and rich media","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.114,"output":1.143}},"doubao-seed-1-6-thinking-250715":{"id":"doubao-seed-1-6-thinking-250715","name":"doubao-seed-1-6-thinking-250715","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":16000},"cost":{"input":0.121,"output":1.21}},"gpt-5":{"id":"gpt-5","name":"gpt-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-08","last_updated":"2025-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"gemini-2.5-flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":2.5}},"grok-4.20-beta-0309-reasoning":{"id":"grok-4.20-beta-0309-reasoning","name":"grok-4.20-beta-0309-reasoning","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":2,"output":6}},"gpt-5.2-chat-latest":{"id":"gpt-5.2-chat-latest","name":"gpt-5.2-chat-latest","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-12","last_updated":"2025-12-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14}},"claude-opus-4-1-20250805-thinking":{"id":"claude-opus-4-1-20250805-thinking","name":"claude-opus-4-1-20250805-thinking","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-27","last_updated":"2025-05-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"cost":{"input":15,"output":75}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10}},"grok-4-fast-non-reasoning":{"id":"grok-4-fast-non-reasoning","name":"grok-4-fast-non-reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":0.2,"output":0.5}},"claude-opus-4-5-20251101":{"id":"claude-opus-4-5-20251101","name":"claude-opus-4-5-20251101","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-11-25","last_updated":"2025-11-25","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.12,"output":0.69}},"o3":{"id":"o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8}},"gpt-5.3-chat-latest":{"id":"gpt-5.3-chat-latest","name":"GPT-5.3 Chat (latest)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":1.75,"output":14}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}}}},"cohere":{"id":"cohere","env":["COHERE_API_KEY"],"npm":"@ai-sdk/cohere","name":"Cohere","doc":"https://docs.cohere.com/docs/models","models":{"command-r7b-arabic-02-2025":{"id":"command-r7b-arabic-02-2025","name":"Command R7B Arabic","description":"Open Command R model optimized for Arabic enterprise chat, RAG, and cultural knowledge","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-02-27","last_updated":"2025-02-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"command-a-plus-05-2026":{"id":"command-a-plus-05-2026","name":"Command A Plus","description":"Cohere's stronger command model for multilingual agents and enterprise workflows","family":"command-a","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04-01","release_date":"2026-05-20","last_updated":"2026-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":64000},"cost":{"input":2.5,"output":10}},"command-a-reasoning-08-2025":{"id":"command-a-reasoning-08-2025","name":"Command A Reasoning","description":"Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows","family":"command-a","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","min":1}],"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":32000},"cost":{"input":2.5,"output":10}},"command-a-vision-07-2025":{"id":"command-a-vision-07-2025","name":"Command A Vision","description":"Cohere vision model for multilingual document analysis, OCR, and image understanding","family":"command-a","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8000},"cost":{"input":2.5,"output":10}},"north-mini-code-1-0":{"id":"north-mini-code-1-0","name":"North Mini Code","description":"Cohere coding model for practical software engineering and agentic edits","family":"north","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-09-23","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"provider":{"npm":"@ai-sdk/openai-compatible","api":"https://api.cohere.ai/compatibility/v1"},"cost":{"input":0,"output":0}},"command-r-plus-08-2024":{"id":"command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"command-a-translate-08-2025":{"id":"command-a-translate-08-2025","name":"Command A Translate","description":"Translation model for multilingual conversion, localization, and cross-language workflows","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-08-28","last_updated":"2025-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":8000},"cost":{"input":2.5,"output":10}},"command-a-03-2025":{"id":"command-a-03-2025","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":8000},"cost":{"input":2.5,"output":10}},"c4ai-aya-expanse-32b":{"id":"c4ai-aya-expanse-32b","name":"Aya Expanse 32B","description":"Open multilingual model optimized for generation across 23 languages","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-24","last_updated":"2024-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000}},"c4ai-aya-expanse-8b":{"id":"c4ai-aya-expanse-8b","name":"Aya Expanse 8B","description":"Compact open multilingual model optimized for generation across 23 languages","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-10-24","last_updated":"2024-10-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":4000}},"c4ai-aya-vision-8b":{"id":"c4ai-aya-vision-8b","name":"Aya Vision 8B","description":"Compact open multilingual vision model for OCR and visual question answering","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-04","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4000}},"command-r7b-12-2024":{"id":"command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"c4ai-aya-vision-32b":{"id":"c4ai-aya-vision-32b","name":"Aya Vision 32B","description":"Open multilingual vision model for OCR, visual reasoning, and image question answering","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2025-03-04","last_updated":"2025-05-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":16000,"output":4000}},"command-r-08-2024":{"id":"command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}}}},"upstage":{"id":"upstage","env":["UPSTAGE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.upstage.ai/v1/solar","name":"Upstage","doc":"https://developers.upstage.ai/docs/apis/chat","models":{"solar-mini":{"id":"solar-mini","name":"solar-mini","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"solar-mini","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-06-12","last_updated":"2025-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":4096},"cost":{"input":0.15,"output":0.15}},"solar-pro4":{"id":"solar-pro4","name":"Solar Pro 4","description":"Upstage's flagship model, specialized for agentic use","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02","release_date":"2026-08-06","last_updated":"2026-08-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":524288,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"solar-pro3":{"id":"solar-pro3","name":"solar-pro3","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.25,"output":0.25}},"solar-pro2":{"id":"solar-pro2","name":"solar-pro2","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"solar-pro","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":8192},"cost":{"input":0.25,"output":0.25}}}},"sarvam":{"id":"sarvam","env":["SARVAM_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.sarvam.ai/v1","name":"Sarvam AI","doc":"https://docs.sarvam.ai/api-reference-docs/getting-started/models","models":{"sarvam-105b":{"id":"sarvam-105b","name":"Sarvam-105B","description":"Flagship Indian-language reasoning model for enterprise multilingual applications","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":[null,"low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-18","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072}},"sarvam-30b":{"id":"sarvam-30b","name":"Sarvam-30B","description":"Efficient Indian-language reasoning model for chat, coding, and multilingual work","family":"sarvam","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":[null,"low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-18","last_updated":"2026-03-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":65536}}}},"xai":{"id":"xai","env":["XAI_API_KEY"],"npm":"@ai-sdk/xai","name":"xAI","doc":"https://docs.x.ai/docs/models","models":{"grok-4.3":{"id":"grok-4.3","name":"Grok 4.3","description":"xAI's Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4.20-0309-reasoning":{"id":"grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-4.20-multi-agent-0309":{"id":"grok-4.20-multi-agent-0309","name":"Grok 4.20 Multi-Agent","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"grok-imagine-image":{"id":"grok-imagine-image","name":"Grok Imagine Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":16000,"output":0}},"grok-imagine-video":{"id":"grok-imagine-video","name":"Grok Imagine Video","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-01-28","last_updated":"2026-01-28","modalities":{"input":["text","image","video","pdf"],"output":["video"]},"open_weights":false,"limit":{"context":1024,"output":0}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"grok-build-0.1":{"id":"grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"grok-imagine-video-1.5":{"id":"grok-imagine-video-1.5","name":"Grok Imagine Video 1.5","description":"Video model for image-to-video generation, editing, and extension workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-05-30","last_updated":"2026-05-30","modalities":{"input":["text","image","audio","pdf"],"output":["video"]},"open_weights":false,"limit":{"context":1024,"output":0}},"grok-imagine-image-2.0":{"id":"grok-imagine-image-2.0","name":"Grok Imagine Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-08-07","last_updated":"2026-08-07","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":64000,"output":0}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"grok-imagine-image-quality":{"id":"grok-imagine-image-quality","name":"Grok Imagine Image Quality","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"grok","attachment":true,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text","image","pdf"],"output":["image","pdf"]},"open_weights":false,"limit":{"context":16000,"output":0}},"grok-4.20-0309-non-reasoning":{"id":"grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}}}},"zenifra":{"id":"zenifra","env":["ZENIFRA_AI_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://ai.zenifra.com/v1","name":"Zenifra","doc":"https://docs.zenifra.com","models":{"alibaba/qwen3.6-35b-a3b":{"id":"alibaba/qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"provider":{"shape":"completions"},"cost":{"input":0.19,"output":0.48}}}},"zai":{"id":"zai","env":["ZHIPU_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.z.ai/api/paas/v4","name":"Z.AI","doc":"https://docs.z.ai/guides/overview/pricing","models":{"glm-4.7":{"id":"glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5-air":{"id":"glm-4.5-air","name":"GLM-4.5-Air","description":"Lighter GLM-4.5 variant for fast coding assistance and cheaper agents","family":"glm-air","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.2,"output":1.1,"cache_read":0.03,"cache_write":0}},"glm-4.6":{"id":"glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.6v":{"id":"glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.3,"output":0.9}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-4.5-flash":{"id":"glm-4.5-flash","name":"GLM-4.5-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.075,"output":0.25,"cache_read":0.015,"cache_write":0}},"glm-4.5":{"id":"glm-4.5","name":"GLM-4.5","description":"Hybrid-reasoning GLM release that made the 4.5 line broadly useful","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":98304},"cost":{"input":0.6,"output":2.2,"cache_read":0.11,"cache_write":0}},"glm-4.5v":{"id":"glm-4.5v","name":"GLM-4.5V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-08-11","last_updated":"2025-08-11","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":16384},"cost":{"input":0.6,"output":1.8}},"glm-4.7-flashx":{"id":"glm-4.7-flashx","name":"GLM-4.7-FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4,"cache_read":0.01,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26,"cache_write":0}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24,"cache_write":0}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"bailing":{"id":"bailing","env":["BAILING_API_TOKEN"],"npm":"@ai-sdk/openai-compatible","api":"https://api.tbox.cn/api/llm/v1/chat/completions","name":"Bailing","doc":"https://alipaytbox.yuque.com/sxs0ba/ling/intro","models":{"Ring-1T":{"id":"Ring-1T","name":"Ring-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"ring","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-06","release_date":"2025-10","last_updated":"2025-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.57,"output":2.29}},"Ling-1T":{"id":"Ling-1T","name":"Ling-1T","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"ling","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-10","last_updated":"2025-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32000},"cost":{"input":0.57,"output":2.29}}}},"tencent-tokenhub":{"id":"tencent-tokenhub","env":["TENCENT_TOKENHUB_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://tokenhub.tencentmaas.com/v1","name":"Tencent TokenHub","doc":"https://cloud.tencent.com/document/product/1823/130050","models":{"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"hy3-preview":{"id":"hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"runinfra":{"id":"runinfra","env":["RUNINFRA_GATEWAY_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.runinfra.ai/v1","name":"RunInfra","doc":"https://runinfra.ai/docs","models":{"ornith-ai/Ornith-1.5-35B-A3B":{"id":"ornith-ai/Ornith-1.5-35B-A3B","name":"Ornith 1.5 35B A3B","description":"Mixture-of-experts coding-reasoning model for agentic software tasks, tool use, and image understanding","family":"ornith","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-18","last_updated":"2026-08-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"Inferact/Qwen3.8-2.4T-A95B-NVFP4":{"id":"Inferact/Qwen3.8-2.4T-A95B-NVFP4","name":"Qwen3.8 2.4T A95B (NVFP4)","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":2,"output":6,"cache_read":0.2}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.13,"output":0.27,"cache_read":0.01}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.6,"output":1.9,"cache_read":0.03}},"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16":{"id":"nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16","name":"Nemotron 3.5 Lightning 30B A3B","description":"Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-11","last_updated":"2026-08-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.05,"output":0.15,"cache_read":0.01}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}},"Qwen/Qwen3.8-27B":{"id":"Qwen/Qwen3.8-27B","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.01}}}},"ai-router":{"id":"ai-router","env":["AI_ROUTER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.ai-router.dev/v1","name":"AI-ROUTER","doc":"https://ai-router.dev/openai-compatible-api-gateway/","models":{"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125}},"gpt-5.5":{"id":"gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}}}},"berget":{"id":"berget","env":["BERGET_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.berget.ai/v1","name":"Berget.AI","doc":"https://api.berget.ai","models":{"mistralai/Mistral-Small-3.2-24B-Instruct-2506":{"id":"mistralai/Mistral-Small-3.2-24B-Instruct-2506","name":"Mistral Small 3.2 24B Instruct 2506","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-10-01","last_updated":"2025-10-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":8192},"cost":{"input":0.33,"output":0.33}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["audio","image","text","video"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":8192},"cost":{"input":0.275,"output":0.55}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":32768},"cost":{"input":1.54,"output":4.84}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-09-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":16384},"cost":{"input":0.29,"output":0.58}},"Qwen/Qwen3.8-27B-FP8":{"id":"Qwen/Qwen3.8-27B-FP8","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.46,"output":3.48}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-27","last_updated":"2026-07-28","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":327680,"output":32768},"cost":{"input":3,"output":15}}}},"mistral":{"id":"mistral","env":["MISTRAL_API_KEY"],"npm":"@ai-sdk/mistral","name":"Mistral","doc":"https://docs.mistral.ai/getting-started/models/","models":{"pixtral-12b":{"id":"pixtral-12b","name":"Pixtral 12B","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-09","release_date":"2024-09-01","last_updated":"2024-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"devstral-small-2507":{"id":"devstral-small-2507","name":"Devstral Small","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"mistral-small-2506":{"id":"mistral-small-2506","name":"Mistral Small 3.2","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-06-20","last_updated":"2025-06-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.1,"output":0.3}},"magistral-small":{"id":"magistral-small","name":"Magistral Small","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-small","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.5,"output":1.5}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral-embed":{"id":"mistral-embed","name":"Mistral Embed","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","family":"mistral-embed","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8000,"output":3072},"cost":{"input":0.1,"output":0}},"devstral-small-2505":{"id":"devstral-small-2505","name":"Devstral Small 2505","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.1,"output":0.3}},"labs-devstral-small-2512":{"id":"labs-devstral-small-2512","name":"Devstral Small 2","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"status":"deprecated","cost":{"input":0,"output":0}},"magistral-medium-latest":{"id":"magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":2,"output":5}},"open-mixtral-8x22b":{"id":"open-mixtral-8x22b","name":"Mixtral 8x22B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-17","last_updated":"2024-04-17","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":64000,"output":64000},"cost":{"input":2,"output":6}},"open-mixtral-8x7b":{"id":"open-mixtral-8x7b","name":"Mixtral 8x7B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mixtral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-01","release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.7,"output":0.7}},"open-mistral-7b":{"id":"open-mistral-7b","name":"Mistral 7B","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-09-27","last_updated":"2023-09-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8000,"output":8000},"cost":{"input":0.25,"output":0.25}},"mistral-medium-latest":{"id":"mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}},"devstral-medium-2507":{"id":"devstral-medium-2507","name":"Devstral Medium","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-07-10","last_updated":"2025-07-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2}},"mistral-medium-2604":{"id":"mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"devstral-medium-latest":{"id":"devstral-medium-latest","name":"Devstral 2 (latest)","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"voxtral-small-latest":{"id":"voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.1,"output":0.3}},"ministral-8b-latest":{"id":"ministral-8b-latest","name":"Ministral 8B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.1,"output":0.1}},"mistral-nemo":{"id":"mistral-nemo","name":"Mistral Nemo","description":"Efficient Mistral-NVIDIA open model for multilingual chat and local deployment","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.15,"output":0.15}},"voxtral-mini-tts-latest":{"id":"voxtral-mini-tts-latest","name":"Voxtral Mini TTS (latest)","description":"Multilingual text-to-speech model with zero-shot voice cloning","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-03-01","last_updated":"2026-03-01","modalities":{"input":["text"],"output":["audio"]},"open_weights":false,"limit":{"context":0,"output":0}},"mistral-small-latest":{"id":"mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"open-mistral-nemo":{"id":"open-mistral-nemo","name":"Open Mistral Nemo","description":"Legacy model retained for compatibility with older integrations","family":"mistral-nemo","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-07-01","last_updated":"2024-07-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"status":"deprecated","cost":{"input":0.15,"output":0.15}},"mistral-large-latest":{"id":"mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5}},"voxtral-mini-latest":{"id":"voxtral-mini-latest","name":"Voxtral Mini (latest)","description":"Speech transcription model for accurate audio-to-text and captioning workflows","family":"voxtral","attachment":false,"reasoning":false,"tool_call":false,"temperature":false,"release_date":"2026-02-01","last_updated":"2026-02-01","modalities":{"input":["audio"],"output":["text"]},"open_weights":false,"limit":{"context":0,"output":0}},"mistral-large-2411":{"id":"mistral-large-2411","name":"Mistral Large 2.1","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-18","last_updated":"2024-11-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":2,"output":6}},"devstral-latest":{"id":"devstral-latest","name":"Devstral 2","description":"Legacy model retained for compatibility with older integrations","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.4,"output":2}},"zai-glm-5-2":{"id":"zai-glm-5-2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"status":"beta","cost":{"input":1.4,"output":4.4,"cache_read":0.14}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.15,"output":0.6}},"mistral-medium-2505":{"id":"mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.4,"output":2}},"codestral-latest":{"id":"codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9}},"ministral-3b-latest":{"id":"ministral-3b-latest","name":"Ministral 3B (latest)","description":"Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads","family":"ministral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-10-01","last_updated":"2024-10-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.04,"output":0.04}},"mistral-medium-2508":{"id":"mistral-medium-2508","name":"Mistral Medium 3.1","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-08-12","last_updated":"2025-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"pixtral-large-latest":{"id":"pixtral-large-latest","name":"Pixtral Large (latest)","description":"Mistral's larger vision model for document-heavy image understanding and chat","family":"pixtral","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":2,"output":6}}}},"synthetic":{"id":"synthetic","env":["SYNTHETIC_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.synthetic.new/openai/v1","name":"Synthetic","doc":"https://synthetic.new/pricing","models":{"hf:openai/gpt-oss-120b":{"id":"hf:openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.1,"output":0.1,"cache_read":0.1}},"hf:MiniMaxAI/MiniMax-M3":{"id":"hf:MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.6,"output":1.2,"cache_read":0.6}},"hf:moonshotai/Kimi-K2.7-Code":{"id":"hf:moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.95}},"hf:moonshotai/Kimi-K3":{"id":"hf:moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-16","last_updated":"2026-07-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":3,"output":15,"cache_read":0.45}},"hf:Qwen/Qwen3.6-27B":{"id":"hf:Qwen/Qwen3.6-27B","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":3.6,"cache_read":0.45}},"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4":{"id":"hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4","name":"Nemotron 3 Super 120B A12B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1,"cache_read":0.3}},"hf:zai-org/GLM-5.2":{"id":"hf:zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","xhigh"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":1.4,"output":4.4,"cache_read":1.4}},"hf:zai-org/GLM-4.7-Flash":{"id":"hf:zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":65536},"cost":{"input":0.1,"output":0.5,"cache_read":0.1}},"hf:zai-org/GLM-5.3-Flash":{"id":"hf:zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":65536},"cost":{"input":0.15,"output":0.5,"cache_read":0.04}}}},"mixlayer":{"id":"mixlayer","env":["MIXLAYER_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://models.mixlayer.ai/v1","name":"Mixlayer","doc":"https://docs.mixlayer.com","models":{"qwen/qwen3.5-9b":{"id":"qwen/qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3.5-27b":{"id":"qwen/qwen3.5-27b","name":"Qwen3.5 27B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":2.4}},"qwen/qwen3.5-35b-a3b":{"id":"qwen/qwen3.5-35b-a3b","name":"Qwen3.5 35B A3B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.25,"output":1.3}},"qwen/qwen3.5-397b-a17b":{"id":"qwen/qwen3.5-397b-a17b","name":"Qwen3.5 397B A17B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3.6}},"qwen/qwen3.5-122b-a10b":{"id":"qwen/qwen3.5-122b-a10b","name":"Qwen3.5 122B A10B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":3.2}}}},"longcat":{"id":"longcat","env":["LONGCAT_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.longcat.chat/openai","name":"LongCat","doc":"https://longcat.chat/platform/docs/","models":{"LongCat-2.0":{"id":"LongCat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.75,"output":2.95,"cache_read":0.015}}}},"cerebras":{"id":"cerebras","env":["CEREBRAS_API_KEY"],"npm":"@ai-sdk/cerebras","name":"Cerebras","doc":"https://inference-docs.cerebras.ai/models/overview","models":{"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2026-06-10","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":40960},"cost":{"input":0.35,"output":0.75}},"qwen-3.8-27b":{"id":"qwen-3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-09-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":32768},"cost":{"input":0.99,"output":1.49}}}},"togetherai":{"id":"togetherai","env":["TOGETHER_API_KEY"],"npm":"@ai-sdk/togetherai","name":"Together AI","doc":"https://docs.together.ai/docs/serverless-models","models":{"essentialai/Rnj-1-Instruct":{"id":"essentialai/Rnj-1-Instruct","name":"Rnj-1 Instruct","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"rnj","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-10","release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"status":"deprecated","cost":{"input":0.15,"output":0.15}},"deepseek-ai/DeepSeek-V3-1":{"id":"deepseek-ai/DeepSeek-V3-1","name":"DeepSeek V3.1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-21","last_updated":"2025-08-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":0.6,"output":1.7}},"deepseek-ai/DeepSeek-V3":{"id":"deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3","description":"Legacy model retained for compatibility with older integrations","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"status":"deprecated","cost":{"input":1.25,"output":1.25}},"deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"deepseek-ai/DeepSeek-R1":{"id":"deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1","description":"Legacy model retained for compatibility with older integrations","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163839,"output":163839},"status":"deprecated","cost":{"input":3,"output":7}},"deepseek-ai/DeepSeek-V4-Pro":{"id":"deepseek-ai/DeepSeek-V4-Pro","name":"DeepSeek V4 Pro","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":384000},"cost":{"input":1.74,"output":3.48,"cache_read":0.2}},"pearl-ai/gemma-4-31b-it":{"id":"pearl-ai/gemma-4-31b-it","name":"Pearl AI Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.28,"output":0.86}},"nvidia/nemotron-3-ultra-550b-a55b":{"id":"nvidia/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512300,"output":512300},"cost":{"input":0.6,"output":3.6,"cache_read":0.2}},"google/gemma-4-31B-it":{"id":"google/gemma-4-31B-it","name":"Gemma 4 31B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.39,"output":0.97}},"google/gemma-3n-E4B-it":{"id":"google/gemma-3n-E4B-it","name":"Gemma 3N E4B Instruct","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-05-20","last_updated":"2025-05-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.06,"output":0.12}},"zai-org/GLM-5.1":{"id":"zai-org/GLM-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-11","release_date":"2026-04-07","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.3":{"id":"zai-org/GLM-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":262144},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5.2":{"id":"zai-org/GLM-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-16","last_updated":"2026-06-16","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":512000,"output":164000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/GLM-5":{"id":"zai-org/GLM-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"status":"deprecated","cost":{"input":1,"output":3.2}},"zai-org/GLM-5.3-Flash":{"id":"zai-org/GLM-5.3-Flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048575,"output":400000},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"LiquidAI/LFM2-24B-A2B":{"id":"LiquidAI/LFM2-24B-A2B","name":"LFM2-24B-A2B","description":"Open-weight instruction model for adaptable chat and self-hosted production workloads","family":"liquid","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-02-25","last_updated":"2026-02-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.03,"output":0.12}},"thinkingmachines/Inkling":{"id":"thinkingmachines/Inkling","name":"Inkling","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max","xhigh","high","medium","low","none"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":131072},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"Qwen/Qwen3.7-Max":{"id":"Qwen/Qwen3.7-Max","name":"Qwen3.7 Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":500000},"cost":{"input":1.25,"output":3.75,"cache_read":0.125}},"Qwen/Qwen3-235B-A22B-Instruct-2507-tput":{"id":"Qwen/Qwen3-235B-A22B-Instruct-2507-tput","name":"Qwen3 235B A22B Instruct 2507 FP8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-07","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.2,"output":0.6}},"Qwen/Qwen3.6-Plus":{"id":"Qwen/Qwen3.6-Plus","name":"Qwen3.6 Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":500000},"cost":{"input":0.5,"output":3}},"Qwen/Qwen3.5-9B":{"id":"Qwen/Qwen3.5-9B","name":"Qwen3.5 9B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.17,"output":0.25}},"Qwen/Qwen3.5-397B-A17B":{"id":"Qwen/Qwen3.5-397B-A17B","name":"Qwen3.5 397B A17B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-02-16","last_updated":"2026-06-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":130000},"status":"deprecated","cost":{"input":0.6,"output":3.6,"cache_read":0.35}},"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8":{"id":"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8","name":"Qwen3 Coder 480B A35B Instruct","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":2,"output":2}},"Qwen/Qwen2.5-7B-Instruct-Turbo":{"id":"Qwen/Qwen2.5-7B-Instruct-Turbo","name":"Qwen 2.5 7B Instruct Turbo","description":"Efficient Qwen model for fast chat, extraction, and high-volume workloads","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2024-09-19","last_updated":"2024-09-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":0.3}},"Qwen/Qwen3-Coder-Next-FP8":{"id":"Qwen/Qwen3-Coder-Next-FP8","name":"Qwen3 Coder Next FP8","description":"Legacy model retained for compatibility with older integrations","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2026-02-03","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.5,"output":1.2}},"deepcogito/cogito-v2-1-671b":{"id":"deepcogito/cogito-v2-1-671b","name":"Cogito v2.1 671B","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","family":"cogito","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":false,"temperature":true,"release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163840,"output":163840},"cost":{"input":1.25,"output":1.25}},"MiniMaxAI/MiniMax-M2.5":{"id":"MiniMaxAI/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"status":"deprecated","cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M3":{"id":"MiniMaxAI/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":250000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"MiniMaxAI/MiniMax-M2.7":{"id":"MiniMaxAI/MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"meta-llama/Llama-3.3-70B-Instruct-Turbo":{"id":"meta-llama/Llama-3.3-70B-Instruct-Turbo","name":"Llama 3.3 70B","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2026-07-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":1.04,"output":1.04}},"meta-llama/Meta-Llama-3-8B-Instruct-Lite":{"id":"meta-llama/Meta-Llama-3-8B-Instruct-Lite","name":"Meta Llama 3 8B Instruct Lite","description":"Compact Llama instruction model for fast chat and local deployment","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2024-04-18","last_updated":"2024-04-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":8192},"cost":{"input":0.14,"output":0.14}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.05,"output":0.2}},"openai/gpt-oss-120b":{"id":"openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.15,"output":0.6}},"moonshotai/Kimi-K2.5":{"id":"moonshotai/Kimi-K2.5","name":"Kimi K2.5","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":true,"temperature":true,"knowledge":"2026-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated","cost":{"input":0.5,"output":2.8}},"moonshotai/Kimi-K2.7-Code":{"id":"moonshotai/Kimi-K2.7-Code","name":"Kimi K2.7 Code","description":"Kimi coding model for software agents, refactors, and repository reasoning","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-14","last_updated":"2026-06-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/Kimi-K2.6":{"id":"moonshotai/Kimi-K2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131000},"cost":{"input":1.2,"output":4.5,"cache_read":0.2}},"moonshotai/Kimi-K3":{"id":"moonshotai/Kimi-K3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}}}},"cloudflare-workers-ai":{"id":"cloudflare-workers-ai","env":["CLOUDFLARE_ACCOUNT_ID","CLOUDFLARE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cloudflare.com/client/v4/accounts/${CLOUDFLARE_ACCOUNT_ID}/ai/v1","name":"Cloudflare Workers AI","doc":"https://developers.cloudflare.com/workers-ai/models/","models":{"@cf/qwen/qwen3-30b-a3b-fp8":{"id":"@cf/qwen/qwen3-30b-a3b-fp8","name":"Qwen3 30B A3b fp8","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-28","last_updated":"2025-04-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.0509,"output":0.335}},"@cf/qwen/qwen3.8-27b":{"id":"@cf/qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":3.2,"cache_read":0.05}},"@cf/qwen/qwq-32b":{"id":"@cf/qwen/qwq-32b","name":"Qwq 32B","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":24000,"output":24000},"cost":{"input":0.66,"output":1}},"@cf/qwen/qwen2.5-coder-32b-instruct":{"id":"@cf/qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5 Coder 32B Instruct","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32768},"cost":{"input":0.66,"output":1}},"@cf/deepseek-ai/deepseek-v4-pro-0813":{"id":"@cf/deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"@cf/deepseek-ai/deepseek-v4-flash-0731":{"id":"@cf/deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b":{"id":"@cf/deepseek-ai/deepseek-r1-distill-qwen-32b","name":"Deepseek R1 Distill Qwen 32B","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":80000,"output":80000},"cost":{"input":0.497,"output":4.881}},"@cf/mistralai/mistral-small-3.1-24b-instruct":{"id":"@cf/mistralai/mistral-small-3.1-24b-instruct","name":"Mistral Small 3.1 24B Instruct","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-18","last_updated":"2025-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"@cf/nvidia/nemotron-3-120b-a12b":{"id":"@cf/nvidia/nemotron-3-120b-a12b","name":"Nemotron 3 Super 120B","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.5,"output":1.5}},"@cf/google/gemma-4-26b-a4b-it":{"id":"@cf/google/gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":16384},"cost":{"input":0.1,"output":0.3}},"@cf/zai-org/glm-5.2":{"id":"@cf/zai-org/glm-5.2","name":"Glm 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"@cf/zai-org/glm-5.3-flash":{"id":"@cf/zai-org/glm-5.3-flash","name":"Glm 5.3 Flash","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1048576},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"@cf/zai-org/glm-5.3":{"id":"@cf/zai-org/glm-5.3","name":"Glm 5.3","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":1310720},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"@cf/zai-org/glm-4.7-flash":{"id":"@cf/zai-org/glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0605,"output":0.4}},"@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"id":"@cf/aisingapore/gemma-sea-lion-v4-27b-it","name":"Gemma Sea Lion V4 27B It","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"@cf/meta/llama-guard-3-8b":{"id":"@cf/meta/llama-guard-3-8b","name":"Llama Guard 3 8B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.484,"output":0.03}},"@cf/meta/llama-3.1-8b-instruct-fp8":{"id":"@cf/meta/llama-3.1-8b-instruct-fp8","name":"Llama 3.1 8B Instruct fp8","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.152,"output":0.287}},"@cf/meta/llama-3.2-3b-instruct":{"id":"@cf/meta/llama-3.2-3b-instruct","name":"Llama 3.2 3B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":80000,"output":80000},"cost":{"input":0.0509,"output":0.335}},"@cf/meta/llama-3.2-1b-instruct":{"id":"@cf/meta/llama-3.2-1b-instruct","name":"Llama 3.2 1B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":60000,"output":60000},"cost":{"input":0.027,"output":0.201}},"@cf/meta/llama-4-scout-17b-16e-instruct":{"id":"@cf/meta/llama-4-scout-17b-16e-instruct","name":"Llama 4 Scout 17B 16E Instruct","description":"Open Llama with long-context vision for efficient multimodal agents","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-04-05","last_updated":"2025-04-05","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":16384},"cost":{"input":0.27,"output":0.85}},"@cf/meta/llama-3.3-70b-instruct-fp8-fast":{"id":"@cf/meta/llama-3.3-70b-instruct-fp8-fast","name":"Llama 3.3 70B Instruct fp8 Fast","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":24000,"output":24000},"cost":{"input":0.293,"output":2.253}},"@cf/meta/llama-3.2-11b-vision-instruct":{"id":"@cf/meta/llama-3.2-11b-vision-instruct","name":"Llama 3.2 11B Vision Instruct","description":"Open Llama multimodal model for image understanding and text reasoning","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.0485,"output":0.676}},"@cf/ibm-granite/granite-4.0-h-micro":{"id":"@cf/ibm-granite/granite-4.0-h-micro","name":"Granite 4.0 H Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"granite","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-07","last_updated":"2025-10-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.017,"output":0.112}},"@cf/openai/gpt-oss-20b":{"id":"@cf/openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.2,"output":0.3}},"@cf/openai/gpt-oss-120b":{"id":"@cf/openai/gpt-oss-120b","name":"GPT OSS 120B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":16384},"cost":{"input":0.35,"output":0.75}},"@cf/moonshotai/kimi-k2.6":{"id":"@cf/moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"@cf/moonshotai/kimi-k2.7-code":{"id":"@cf/moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}}}},"moark":{"id":"moark","env":["MOARK_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://moark.com/v1","name":"Moark","doc":"https://moark.com/docs/openapi/v1#tag/%E6%96%87%E6%9C%AC%E7%94%9F%E6%88%90","models":{"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":2.1,"output":8.4,"cache_read":2.1,"cache_write":8.4}},"GLM-4.7":{"id":"GLM-4.7","name":"GLM-4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":3.5,"output":14}}}},"zenmux":{"id":"zenmux","env":["ZENMUX_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://zenmux.ai/api/v1","name":"ZenMux","doc":"https://docs.zenmux.ai","models":{"qwen/qwen3.7-max":{"id":"qwen/qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3-Coder-Plus","description":"Qwen coding model for software agents, repository edits, and code reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"qwen/qwen3.6-plus":{"id":"qwen/qwen3.6-plus","name":"Qwen3.6-Plus","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"qwen/qwen3.5-flash":{"id":"qwen/qwen3.5-flash","name":"Qwen3.5 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1020000,"output":1020000},"cost":{"input":0.1,"output":0.4}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3-Max-Thinking","description":"Qwen reasoning model for deliberate problem solving, math, and coding","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":1.2,"output":6}},"qwen/qwen3.7-plus":{"id":"qwen/qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.4,"output":1.6,"cache_read":0.08,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.24,"cache_write":1.5}}},"qwen/qwen3.5-plus":{"id":"qwen/qwen3.5-plus","name":"Qwen3.5 Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0.8,"output":4.8}},"baidu/ernie-5.0-thinking-preview":{"id":"baidu/ernie-5.0-thinking-preview","name":"ERNIE 5.0","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.84,"output":3.37}},"volcengine/doubao-seed-code":{"id":"volcengine/doubao-seed-code","name":"Doubao-Seed-Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-11","last_updated":"2025-11-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0.17,"output":1.12,"cache_read":0.03}},"volcengine/doubao-seed-2.0-mini":{"id":"volcengine/doubao-seed-2.0-mini","name":"Doubao-Seed-2.0-mini","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.03,"output":0.28,"cache_read":0.01,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-code":{"id":"volcengine/doubao-seed-2.0-code","name":"Doubao Seed 2.0 Code","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.9,"output":4.48}},"volcengine/doubao-seed-2.0-pro":{"id":"volcengine/doubao-seed-2.0-pro","name":"Doubao-Seed-2.0-pro","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.45,"output":2.24,"cache_read":0.09,"cache_write":0.0024}},"volcengine/doubao-seed-1.8":{"id":"volcengine/doubao-seed-1.8","name":"Doubao-Seed-1.8","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-18","last_updated":"2025-12-18","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.11,"output":0.28,"cache_read":0.02,"cache_write":0.0024}},"volcengine/doubao-seed-2.0-lite":{"id":"volcengine/doubao-seed-2.0-lite","name":"Doubao-Seed-2.0-lite","description":"Multimodal reasoning model for visual analysis, planning, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-14","release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.09,"output":0.51,"cache_read":0.02,"cache_write":0.0024}},"stepfun/step-3.7-flash":{"id":"stepfun/step-3.7-flash","name":"Step 3.7 Flash","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0.2,"output":1.15}},"stepfun/step-3":{"id":"stepfun/step-3","name":"Step-3","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-31","last_updated":"2025-07-31","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":65536,"output":64000},"status":"deprecated","cost":{"input":0.21,"output":0.57}},"stepfun/step-3.5-flash":{"id":"stepfun/step-3.5-flash","name":"Step 3.5 Flash","description":"StepFun flash model for efficient multimodal reasoning, coding, and tool use","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"cost":{"input":0.1,"output":0.3}},"stepfun/step-3.7-flash-free":{"id":"stepfun/step-3.7-flash-free","name":"Step 3.7 Flash (Free)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":256000,"output":256000},"cost":{"input":0,"output":0}},"xiaomi/mimo-v2-flash":{"id":"xiaomi/mimo-v2-flash","name":"MiMo-V2-Flash","description":"MiMo flash model for fast multimodal assistance and agent workflows","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12-01","release_date":"2025-12-16","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.1,"output":0.3,"cache_read":0.01}},"xiaomi/mimo-v2-pro":{"id":"xiaomi/mimo-v2-pro","name":"MiMo V2 Pro","description":"Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":256000},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"xiaomi/mimo-v2-omni":{"id":"xiaomi/mimo-v2-omni","name":"MiMo V2 Omni","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":265000,"output":265000},"cost":{"input":0.4,"output":2,"cache_read":0.08}},"xiaomi/mimo-v2.5":{"id":"xiaomi/mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":2,"cache_read":0.08,"tiers":[{"input":0.8,"output":4,"cache_read":0.16,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":0.8,"output":4,"cache_read":0.16}}},"xiaomi/mimo-v2.5-pro":{"id":"xiaomi/mimo-v2.5-pro","name":"MiMo-V2.5-Pro","description":"Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution","family":"mimo","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"minimax/minimax-m2.1":{"id":"minimax/minimax-m2.1","name":"MiniMax M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.38}},"minimax/minimax-m2":{"id":"minimax/minimax-m2","name":"MiniMax M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.38}},"minimax/minimax-m2.7-highspeed":{"id":"minimax/minimax-m2.7-highspeed","name":"MiniMax M2.7 highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131070},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.611,"output":2.4439}},"minimax/minimax-m2.7":{"id":"minimax/minimax-m2.7","name":"MiniMax M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131070},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3055,"output":1.2219}},"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"minimax/minimax-m3":{"id":"minimax/minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.6,"output":2.4}},"minimax/minimax-m2.5-lightning":{"id":"minimax/minimax-m2.5-lightning","name":"MiniMax M2.5 highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.6,"output":4.8,"cache_read":0.06,"cache_write":0.75}},"anthropic/claude-sonnet-5-free":{"id":"anthropic/claude-sonnet-5-free","name":"Claude Sonnet 5 (Free)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"anthropic/claude-opus-4.8":{"id":"anthropic/claude-opus-4.8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4.1":{"id":"anthropic/claude-opus-4.1","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.6":{"id":"anthropic/claude-sonnet-4.6","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-18","last_updated":"2026-02-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-haiku-4.5":{"id":"anthropic/claude-haiku-4.5","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"anthropic/claude-opus-4.6":{"id":"anthropic/claude-opus-4.6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-06","last_updated":"2026-02-06","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-3.5-haiku":{"id":"anthropic/claude-3.5-haiku","name":"Claude 3.5 Haiku","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2024-11-04","last_updated":"2024-11-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":0.8,"output":4,"cache_read":0.08,"cache_write":1}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-opus-4":{"id":"anthropic/claude-opus-4","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"anthropic/claude-sonnet-4.5":{"id":"anthropic/claude-sonnet-4.5","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-3.7-sonnet":{"id":"anthropic/claude-3.7-sonnet","name":"Claude 3.7 Sonnet","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-02-24","last_updated":"2025-02-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-opus-4.5":{"id":"anthropic/claude-opus-4.5","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-4":{"id":"anthropic/claude-sonnet-4","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"provider":{"npm":"@ai-sdk/anthropic","api":"https://zenmux.ai/api/anthropic/v1"},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":4}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-02-19","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","pdf","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":4.5}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-22","last_updated":"2025-07-22","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.1,"output":0.4,"cache_read":0.03,"cache_write":1}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"cache_read":0.025}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"cache_read":0.15}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":65530},"cost":{"input":0.25,"output":1.5}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":1}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["pdf","image","text","audio","video"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":1.25,"output":10,"cache_read":0.31,"cache_write":4.5}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["pdf","image","text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048000,"output":64000},"cost":{"input":0.3,"output":2.5,"cache_read":0.07,"cache_write":1}},"sapiens-ai/agnes-1.5-lite":{"id":"sapiens-ai/agnes-1.5-lite","name":"Agnes 1.5 Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-26","last_updated":"2026-03-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.12,"output":0.6}},"sapiens-ai/agnes-1.5-pro":{"id":"sapiens-ai/agnes-1.5-pro","name":"Agnes 1.5 Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-21","last_updated":"2026-03-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.16,"output":0.8}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek-V3.2 (Non-thinking Mode)","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.28,"output":0.42,"cache_read":0.03}},"deepseek/deepseek-v3.2":{"id":"deepseek/deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-05","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.28,"output":0.43}},"deepseek/deepseek-v3.2-exp":{"id":"deepseek/deepseek-v3.2-exp","name":"DeepSeek-V3.2-Exp","description":"DeepSeek chat model for instruction following, coding, and analysis","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":163000,"output":64000},"cost":{"input":0.22,"output":0.33}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"kuaishou/kat-coder-pro-v2":{"id":"kuaishou/kat-coder-pro-v2","name":"KAT-Coder-Pro-V2","description":"Coding model for repository understanding, refactors, and agentic engineering tasks","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-30","last_updated":"2026-03-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":80000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"inclusionai/ring-2.6-1t":{"id":"inclusionai/ring-2.6-1t","name":"inclusionAI: Ring-2.6-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-12-31","release_date":"2026-05-07","last_updated":"2026-05-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":65000},"cost":{"input":0.3,"output":2.5,"cache_read":0.06}},"inclusionai/ring-1t":{"id":"inclusionai/ring-1t","name":"Ring-1T","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-12","last_updated":"2025-10-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.56,"output":2.24,"cache_read":0.11}},"inclusionai/ling-1t":{"id":"inclusionai/ling-1t","name":"Ling-1T","description":"Tool-capable chat model for instruction following and agentic application workflows","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-10-09","last_updated":"2025-10-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","cost":{"input":0.56,"output":2.24,"cache_read":0.11}},"x-ai/grok-4.3":{"id":"x-ai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":1000000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"cache_write":0,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"cache_write":0,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4,"cache_write":0}}},"x-ai/grok-code-fast-1":{"id":"x-ai/grok-code-fast-1","name":"Grok Code Fast 1","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":0.2,"output":1.5,"cache_read":0.02}},"x-ai/grok-4":{"id":"x-ai/grok-4","name":"Grok 4","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-09","last_updated":"2025-07-09","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.75}},"x-ai/grok-4.5":{"id":"x-ai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"x-ai/grok-build-0.1":{"id":"x-ai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2}},"x-ai/grok-4.2-fast-non-reasoning":{"id":"x-ai/grok-4.2-fast-non-reasoning","name":"Grok 4.2 Fast Non Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":3,"output":9}},"x-ai/grok-4.1-fast-non-reasoning":{"id":"x-ai/grok-4.1-fast-non-reasoning","name":"Grok 4.1 Fast Non Reasoning","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":64000},"status":"deprecated","cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"x-ai/grok-4.2-fast":{"id":"x-ai/grok-4.2-fast","name":"Grok 4.2 Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":30000},"cost":{"input":3,"output":9}},"x-ai/grok-4-fast":{"id":"x-ai/grok-4-fast","name":"Grok 4 Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-19","last_updated":"2025-09-19","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":64000},"status":"deprecated","cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"x-ai/grok-4.1-fast":{"id":"x-ai/grok-4.1-fast","name":"Grok 4.1 Fast","description":"Fast Grok model for responsive chat, reasoning, and tool-assisted work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":2000000,"output":64000},"status":"deprecated","cost":{"input":0.2,"output":0.5,"cache_read":0.05}},"openai/gpt-5-codex":{"id":"openai/gpt-5-codex","name":"GPT-5 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.1-codex-mini":{"id":"openai/gpt-5.1-codex-mini","name":"GPT-5.1-Codex-Mini","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.25,"output":2,"cache_read":0.03}},"openai/gpt-5.1-codex":{"id":"openai/gpt-5.1-codex","name":"GPT-5.1-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":45,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1,"cache_write":12.5}}},"openai/gpt-5.2-codex":{"id":"openai/gpt-5.2-codex","name":"GPT-5.2-Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-01-01","release_date":"2026-01-15","last_updated":"2026-01-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.17}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2-Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":21,"output":168}},"openai/gpt-5.1-chat":{"id":"openai/gpt-5.1-chat","name":"GPT-5.1 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":3.75,"output":18.75}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":1,"output":6,"cache_read":0.1,"cache_write":1.25,"tiers":[{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":2,"output":9,"cache_read":0.2,"cache_write":2.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 Nano","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.2,"output":1.25}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"tiers":[{"input":60,"output":270,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 Mini","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":0.75,"output":4.5}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"Frontier GPT model for professional reasoning, coding, and multimodal work","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":45,"output":225}},"openai/gpt-5.5-instant":{"id":"openai/gpt-5.5-instant","name":"GPT-5.5 Instant","description":"Compact GPT model for low-latency assistance and high-volume workloads","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-05-05","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":400000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"cache_write":3.125,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5,"cache_write":6.25}}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2025-01-01","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14,"cache_read":0.17}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"output":64000},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.25,"output":10,"cache_read":0.12}},"openai/gpt-5.3-chat":{"id":"openai/gpt-5.3-chat","name":"GPT-5.3 Chat","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16380},"provider":{"npm":"@ai-sdk/openai","api":"https://zenmux.ai/api/v1"},"cost":{"input":1.75,"output":14}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":12.5,"output":75,"cache_read":1.25},"provider":{"body":{"service_tier":"priority"}}}}},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"moonshotai/kimi-k2-0905":{"id":"moonshotai/kimi-k2-0905","name":"Kimi K2 0905","description":"Kimi model for long-context chat, coding, and agentic reasoning","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-04","last_updated":"2025-09-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k2.6":{"id":"moonshotai/kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2025-01-01","release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262140,"output":262140},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2.7-code-free":{"id":"moonshotai/kimi-k2.7-code-free","name":"Kimi K2.7 Code (Free)","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshotai/kimi-k2-thinking":{"id":"moonshotai/kimi-k2-thinking","name":"Kimi K2 Thinking","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":0.6,"output":2.5,"cache_read":0.15}},"moonshotai/kimi-k3":{"id":"moonshotai/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":false,"knowledge":"2025-01-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"cost":{"input":0.58,"output":3.02,"cache_read":0.1}},"moonshotai/kimi-k3-free":{"id":"moonshotai/kimi-k3-free","name":"Kimi K3 (Free)","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"moonshotai/kimi-k2-thinking-turbo":{"id":"moonshotai/kimi-k2-thinking-turbo","name":"Kimi K2 Thinking Turbo","description":"Kimi reasoning model for long-horizon research, planning, and tool use","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":64000},"status":"deprecated","cost":{"input":1.15,"output":8,"cache_read":0.15}},"tencent/hy3-preview":{"id":"tencent/hy3-preview","name":"Hy3 preview","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-04-20","last_updated":"2026-04-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":64000},"cost":{"input":0.172,"output":0.572,"cache_read":0.058,"cache_write":0}},"z-ai/glm-4.7":{"id":"z-ai/glm-4.7","name":"GLM 4.7","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.28,"output":1.14,"cache_read":0.06}},"z-ai/glm-4.5-air":{"id":"z-ai/glm-4.5-air","name":"GLM 4.5 Air","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.11,"output":0.56,"cache_read":0.02}},"z-ai/glm-4.6":{"id":"z-ai/glm-4.6","name":"GLM 4.6","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.35,"output":1.54,"cache_read":0.07}},"z-ai/glm-4.6v":{"id":"z-ai/glm-4.6v","name":"GLM 4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.14,"output":0.42,"cache_read":0.03}},"z-ai/glm-5.2":{"id":"z-ai/glm-5.2","name":"GLM 5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.5,"cache_read":0.26}},"z-ai/glm-5.2-free":{"id":"z-ai/glm-5.2-free","name":"GLM 5.2 (Free)","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0}},"z-ai/glm-4.6v-flash-free":{"id":"z-ai/glm-4.6v-flash-free","name":"GLM 4.6V Flash (Free)","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0}},"z-ai/glm-4.5":{"id":"z-ai/glm-4.5","name":"GLM 4.5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-07-25","last_updated":"2025-07-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":64000},"cost":{"input":0.35,"output":1.54,"cache_read":0.07}},"z-ai/glm-4.7-flashx":{"id":"z-ai/glm-4.7-flashx","name":"GLM 4.7 FlashX","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.07,"output":0.42,"cache_read":0.01}},"z-ai/glm-5":{"id":"z-ai/glm-5","name":"GLM 5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":128000},"cost":{"input":0.58,"output":2.6,"cache_read":0.14}},"z-ai/glm-5.1":{"id":"z-ai/glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-03","last_updated":"2026-04-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":131072},"cost":{"input":0.8781,"output":3.5126,"cache_read":0.1903}},"z-ai/glm-5-turbo":{"id":"z-ai/glm-5-turbo","name":"GLM 5 Turbo","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2026-03-20","last_updated":"2026-03-20","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.88,"output":3.48}},"z-ai/glm-5v-turbo":{"id":"z-ai/glm-5v-turbo","name":"GLM 5V Turbo","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":128000},"cost":{"input":0.726,"output":3.1946,"cache_read":0.1743}},"z-ai/glm-4.7-flash-free":{"id":"z-ai/glm-4.7-flash-free","name":"GLM 4.7 Flash (Free)","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01-01","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0}},"z-ai/glm-4.6v-flash":{"id":"z-ai/glm-4.6v-flash","name":"GLM 4.6V FlashX","description":"GLM vision model for visual reasoning, documents, and multimodal agents","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01-01","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.02,"output":0.21,"cache_read":0.0043}}}},"vancine":{"id":"vancine","env":["VANCINE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://vancine.com/v1","name":"Vancine","doc":"https://vancine.com/docs","models":{"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.67,"output":2,"cache_read":0.034}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":2.4,"output":12,"cache_read":0.24}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.06,"output":0.2,"cache_read":0.012}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.12,"output":0.38,"cache_read":0.013}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":1.6,"output":4.8,"cache_read":0.2}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.24,"output":0.96,"cache_read":0.048}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.12,"output":3.52,"cache_read":0.208}}}},"minimax-cn":{"id":"minimax-cn","env":["MINIMAX_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.minimaxi.com/anthropic/v1","name":"MiniMax (minimaxi.com)","doc":"https://platform.minimaxi.com/docs/guides/quickstart","models":{"MiniMax-M2":{"id":"MiniMax-M2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"MiniMax-M2.1":{"id":"MiniMax-M2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03,"cache_write":0.375}},"MiniMax-M2.5-highspeed":{"id":"MiniMax-M2.5-highspeed","name":"MiniMax-M2.5-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-13","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M3":{"id":"MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-25","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"MiniMax-M2.7-highspeed":{"id":"MiniMax-M2.7-highspeed","name":"MiniMax-M2.7-highspeed","description":"High-speed MiniMax model for low-latency coding and agent workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.6,"output":2.4,"cache_read":0.06,"cache_write":0.375}},"MiniMax-M2.7":{"id":"MiniMax-M2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"cache_write":0.375}}}},"cortecs":{"id":"cortecs","env":["CORTECS_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.cortecs.ai/v1","name":"Cortecs","doc":"https://api.cortecs.ai/v1/models","models":{"ministral-14b-2512":{"id":"ministral-14b-2512","name":"ministral-14b-2512","description":"Ministral 3 14B is a frontier-level 14B multimodal model optimized for local deployment, delivering state-of-the-art text and vision reasoning with a 256K context window and strong agentic capabilities.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.223,"output":0.223,"cache_read":0.022}},"gpt-5-nano":{"id":"gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.06,"output":0.439,"cache_read":0.019}},"qwen3guard-gen-0.6b":{"id":"qwen3guard-gen-0.6b","name":"qwen3guard-gen-0.6b","description":"Qwen3Guard-Gen-0.6B is a lightweight multilingual safety moderation model that classifies prompts and responses into safe, controversial, or unsafe categories.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0,"output":0}},"gemma-4-26b-a4b-it":{"id":"gemma-4-26b-a4b-it","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":81920},"cost":{"input":0.111,"output":0.557}},"nova-2-lite":{"id":"nova-2-lite","name":"Nova 2 Lite","description":"Nova 2 Lite is an advanced multimodal reasoning model that combines efficiency and performance, delivering reliable AI for agentic workflows and enterprise applications.","family":"nova","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65535},"cost":{"input":0.373,"output":3.144}},"mistral-small-2503":{"id":"mistral-small-2503","name":"mistral-small-2503","description":"Combines advanced text and vision capabilities with 24 billion parameters, supporting multilingual tasks and long contexts up to 131k tokens, making it versatile for various applications without sacrificing performance.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-03-20","last_updated":"2025-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.111,"output":0.334}},"mistral-7b-instruct-v0.2":{"id":"mistral-7b-instruct-v0.2","name":"mistral-7b-instruct-v0.2","description":"Mistral 7B Instruct is a compact, 7B parameter model optimized for fast and efficient text and code generation with a 32K token context window.","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":0.159,"output":0.219}},"codestral-2508":{"id":"codestral-2508","name":"Codestral 2508","description":"Mistral coding model for code completion, generation, and developer workflows","family":"mistral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.334,"output":1.003,"cache_read":0.033}},"llama-3.1-8b-instruct":{"id":"llama-3.1-8b-instruct","name":"Llama-3.1-8B-Instruct","description":"Optimized for dialogue, this LLM by Meta outperforms other open-source chat models in benchmarks while prioritizing helpfulness and safety.","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.167,"output":0.167}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":64000},"cost":{"input":2,"output":3.999,"cache_read":0.5}},"gpt-4.1-nano":{"id":"gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.111,"output":0.434,"cache_read":0.056}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.13,"output":0.28,"cache_read":0.03}},"qwen3.8-flash-next":{"id":"qwen3.8-flash-next","name":"Qwen3.8 Flash Next","description":"Open-weight experimental preview of the Qwen4 architecture: hybrid-attention MoE (125B total, 6B active) with vision encoder for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-27","last_updated":"2026-08-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":64000},"cost":{"input":0.201,"output":0.5,"cache_read":0.05}},"minimax-m2.1":{"id":"minimax-m2.1","name":"MiniMax-M2.1","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":196000},"cost":{"input":0.359,"output":1.435}},"claude-4-5-sonnet":{"id":"claude-4-5-sonnet","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":2.989,"output":14.945,"cache_read":0.326,"cache_write":4.078}},"qwen3.5-9b":{"id":"qwen3.5-9b","name":"Qwen3.5 9B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.111,"output":0.167}},"devstral-2512":{"id":"devstral-2512","name":"Devstral 2","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.478,"output":2.392,"cache_read":0.045}},"minimax-m2":{"id":"minimax-m2","name":"MiniMax-M2","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":400000,"output":196000},"cost":{"input":0.349,"output":1.405}},"gpt-5.6-sol":{"id":"gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5.5,"output":32.998,"cache_read":0.55,"cache_write":6.879}},"qwen3.8-27b":{"id":"qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.334,"output":2.451,"cache_read":0.111}},"claude-opus-5":{"id":"claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.5,"output":27.498,"cache_read":0.55,"cache_write":6.874}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"output":196608},"cost":{"input":0.668,"output":2.674}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.773,"output":3.38,"cache_read":0.193}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1000000},"cost":{"input":1.2,"output":4.2,"cache_read":0.26}},"claude-opus4-5":{"id":"claude-opus4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5.313,"output":26.568,"cache_read":0.531,"cache_write":6.645}},"claude-opus4-6":{"id":"claude-opus4-6","name":"Claude Opus 4.6","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.313,"output":26.561,"cache_read":0.531,"cache_write":6.645}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":196000},"cost":{"input":0.296,"output":1.186,"cache_read":0.075}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.395,"output":1.977,"cache_read":0.099}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.75,"output":3.5,"cache_read":0.201}},"llama-3.1-405b-instruct":{"id":"llama-3.1-405b-instruct","name":"Llama 3.1 405B Instruct","description":"Open Llama instruction model for multilingual chat, reasoning, and coding","family":"llama","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":1.95,"output":1.95}},"qwen3-32b":{"id":"qwen3-32b","name":"Qwen3 32B","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":32000},"cost":{"input":0.089,"output":0.312}},"apertus-70b":{"id":"apertus-70b","name":"Apertus 70B","description":"Apertus 70B is an open, multilingual language model designed for research, long-context reasoning, and sovereignty-focused AI systems.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-09","release_date":"2025-09-02","last_updated":"2025-09-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":65536,"output":16384},"cost":{"input":1.393,"output":2.228}},"gpt-4.1-mini":{"id":"gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.434,"output":1.704,"cache_read":0.134}},"gemini-3.6-flash":{"id":"gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.038}},"gpt-5.4":{"id":"gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"output":128000},"cost":{"input":2.898,"output":15.453,"cache_read":0.242}},"gpt-oss-20b":{"id":"gpt-oss-20b","name":"GPT OSS 20B","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.045,"output":0.167}},"gemini-3.1-flash-lite":{"id":"gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.272,"output":1.631,"cache_read":0.025,"cache_write":0.082}},"mistral-large-2402":{"id":"mistral-large-2402","name":"mistral-large-2402","description":"Mistral Large (24.02) is Mistral AI’s most advanced language model, built for complex multilingual reasoning, code generation, and deep text understanding.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":8192},"cost":{"input":4.284,"output":12.952}},"qwen2.5-vl-72b-instruct":{"id":"qwen2.5-vl-72b-instruct","name":"qwen2.5-vl-72b-instruct","description":"Qwen2.5-VL is a powerful vision-language model with advanced capabilities in visual understanding, long video reasoning, and structured output generation.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2025-01-27","last_updated":"2025-01-27","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":1.014,"output":1.014}},"qwen3-coder-next":{"id":"qwen3-coder-next","name":"Qwen3 Coder Next","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.167,"output":0.891}},"claude-opus4-7":{"id":"claude-opus4-7","name":"Claude Opus 4.7","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.437,"output":27.186,"cache_read":0.544,"cache_write":6.797}},"qwen3-coder-30b-a3b-instruct":{"id":"qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.067,"output":0.245,"cache_read":0.014}},"gpt-5.1":{"id":"gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":10.96,"cache_read":0.156}},"qwen3.5-397b-a17b":{"id":"qwen3.5-397b-a17b","name":"Qwen3.5 397B-A17B","description":"Large open Qwen multimodal MoE for visual agents and long technical tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-15","last_updated":"2026-02-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.668,"output":4.01}},"gpt-oss-safeguard-120b":{"id":"gpt-oss-safeguard-120b","name":"GPT OSS Safeguard 120B","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.179,"output":0.697}},"mistral-large-2512":{"id":"mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.557,"output":1.671,"cache_read":0.056}},"gemini-3.5-flash":{"id":"gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.649,"output":9.899,"cache_read":0.165,"cache_write":1}},"qwen3.6-27b":{"id":"qwen3.6-27b","name":"Qwen3.6 27B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.446,"output":3.008}},"gpt-4o":{"id":"gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":2.659,"output":10.635,"cache_read":1.33}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.219,"output":1.32,"cache_read":0.022,"cache_write":0.275}},"ministral-3b-2512":{"id":"ministral-3b-2512","name":"ministral-3b-2512","description":"Ministral 3 3B is a compact, efficient multimodal model with strong language, vision capabilities, and ideal for custom fine-tuning.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.111,"output":0.111,"cache_read":0.011}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":3,"output":14.999}},"gemma-3-27b-it":{"id":"gemma-3-27b-it","name":"gemma-3-27b-it","description":"Gemma 3 is a family of lightweight, multimodal models from Google, supporting text and image inputs, multilingual capabilities, and a 131K context window.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":false,"release_date":"2025-03-12","last_updated":"2025-03-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":110000},"cost":{"input":0.099,"output":0.299}},"deepseek-r1-0528":{"id":"deepseek-r1-0528","name":"DeepSeek R1 0528","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-05-28","last_updated":"2025-05-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":164000,"output":164000},"cost":{"input":0.652,"output":2.57,"cache_read":0.163}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-07","release_date":"2025-12-01","last_updated":"2025-12-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.296,"output":0.495,"cache_read":0.075}},"qwen3.6-35b-a3b":{"id":"qwen3.6-35b-a3b","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":32768},"cost":{"input":0.167,"output":0.557}},"mistral-nemo-instruct-2407":{"id":"mistral-nemo-instruct-2407","name":"mistral-nemo-instruct-2407","description":"A 12B parameter, instruct-tuned language model by Mistral AI and NVIDIA, designed for advanced instruction following, multi-turn conversations, and generating text and code across multiple languages.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-08-07","last_updated":"2024-08-07","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.145,"output":0.145,"cache_read":0.014}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":0.201,"output":0.5,"cache_read":0.05}},"gpt-4o-mini":{"id":"gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16000},"cost":{"input":0.159,"output":0.638,"cache_read":0.081}},"gemini-3.5-flash-lite":{"id":"gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.33,"output":2.749,"cache_read":0.033}},"qwen3.8-2.4t-a95b":{"id":"qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2.5,"output":6,"cache_read":0.625}},"gpt-4.1":{"id":"gpt-4.1","name":"GPT-4.1","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2.192,"output":8.769,"cache_read":0.546}},"qwen3.5-122b-a10b":{"id":"qwen3.5-122b-a10b","name":"Qwen3.5 122B-A10B","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2026-02-23","last_updated":"2026-02-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.495,"output":3.46,"cache_read":0.124}},"claude-opus4-8":{"id":"claude-opus4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5.437,"output":27.186,"cache_read":0.544,"cache_write":6.797}},"pixtral-large-2502":{"id":"pixtral-large-2502","name":"Pixtral Large (25.02)","description":"Pixtral Large (25.02) is a 124B open-weight multimodal model built on Mistral Large 2, offering advanced image understanding and strong performance across text and code tasks.","family":"pixtral","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":1.993,"output":5.978}},"nova-micro-v1":{"id":"nova-micro-v1","name":"nova-micro-v1","description":"Nova Micro is a multilingual text-to-text foundation model with strong reasoning capabilities and broad language coverage across 200+ languages.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":10000},"cost":{"input":0.04,"output":0.159}},"qwen3-30b-a3b-instruct-2507":{"id":"qwen3-30b-a3b-instruct-2507","name":"qwen3-30b-a3b-instruct-2507","description":"Qwen3-30B-A3B-Instruct-2507 is an advanced Mixture-of-Experts model optimized for reasoning, coding, and multilingual instruction following.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262000,"output":262000},"cost":{"input":0.099,"output":0.299}},"mistral-small-3.2-24b-instruct-2506":{"id":"mistral-small-3.2-24b-instruct-2506","name":"mistral-small-3.2-24b-instruct-2506","description":"Mistral-Small-3.2-24B-Instruct-2506 is a 24B parameter instruction-tuned model with enhanced long-context support (128k) and state-of-the-art vision understanding.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131000,"output":131000},"cost":{"input":0.1,"output":0.312}},"mistral-7b-instruct-v0.3":{"id":"mistral-7b-instruct-v0.3","name":"mistral-7b-instruct-v0.3","description":"Mistral-7B-Instruct-v0.3 model is a fine-tuned version of the Mistral 7B base model, optimized for instruction-following tasks. Released in 2023, it is intended for demonstration purposes and does not include built-in guardrails or moderation features.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-05-26","last_updated":"2025-05-26","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":127000,"output":127000},"cost":{"input":0.111,"output":0.111}},"nova-pro-v1":{"id":"nova-pro-v1","name":"Nova Pro 1.0","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.918,"output":3.671}},"mixtral-8x7B-instruct-v0.1":{"id":"mixtral-8x7B-instruct-v0.1","name":"Mixtral 8x7B Instruct v0.1","description":"Reasoning model for deliberate analysis, multi-step problem solving, and tool use","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2023-12-11","last_updated":"2023-12-11","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":32000,"output":4096},"cost":{"input":0.488,"output":0.758}},"gemma-4-31b-it":{"id":"gemma-4-31b-it","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.223,"output":0.39}},"claude-haiku-4-5":{"id":"claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0.996,"output":4.982,"cache_read":0.099,"cache_write":1.186}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":0.988,"output":3.164,"cache_read":0.247}},"qwen3guard-gen-8b":{"id":"qwen3guard-gen-8b","name":"qwen3guard-gen-8b","description":"Qwen3Guard-Gen-8B is a large-scale multilingual safety moderation model designed for high-accuracy prompt and response classification.","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2026-02-04","last_updated":"2026-02-04","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0,"output":0}},"nvidia-nemotron-3-nano-30b-a3b":{"id":"nvidia-nemotron-3-nano-30b-a3b","name":"nvidia-nemotron-3-nano-30b-a3b","description":"Nemotron-Nano-3-30B-A3B is a compact Mixture-of-Experts model optimized for efficient reasoning, chat, and coding, with strong multilingual support and long-context RAG and agent workflows.","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-01-12","last_updated":"2026-01-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.06,"output":0.24}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Kimi reasoning model for long-horizon research, planning, and tool use","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.495,"output":2.768,"cache_read":0.124}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":202752},"cost":{"input":1.384,"output":4.348,"cache_read":0.346}},"hermes-4-405b":{"id":"hermes-4-405b","name":"hermes-4-405b","description":"Hermes 4 405B is a frontier hybrid-mode reasoning model built on Llama 3.1, optimized for advanced logic, math, coding, and structured output generation.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-08-13","last_updated":"2024-08-13","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.996,"output":2.989}},"qwen3-235b-a22b-instruct-2507":{"id":"qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262000,"output":262000},"cost":{"input":0.069,"output":0.455,"cache_read":0.018}},"mistral-small-2603":{"id":"mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.143,"output":0.568,"cache_read":0.014}},"gemini-3.8-flash":{"id":"gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.825,"output":4.125,"cache_read":0.082,"cache_write":0.084}},"claude-4-6-sonnet":{"id":"claude-4-6-sonnet","name":"Claude Sonnet 4.6","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3.196,"output":15.94,"cache_read":0.32,"cache_write":3.999}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.73,"output":3.46,"cache_read":0.432}},"ministral-8b-2512":{"id":"ministral-8b-2512","name":"ministral-8b-2512","description":"Ministral 3 8B is a balanced, efficient multimodal model offering strong text and vision capabilities, optimized for edge and local deployment.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-12-03","last_updated":"2025-12-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.167,"output":0.167,"cache_read":0.017}},"glm-5-turbo":{"id":"glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":202752},"cost":{"input":1.186,"output":3.955,"cache_read":0.296,"cache_write":1.544}},"gpt-5-mini":{"id":"gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.279,"output":2.192,"cache_read":0.056}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"GPT OSS 120B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.089,"output":0.446,"cache_read":0.01}},"gemini-3.7-flash":{"id":"gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.75,"output":3.75,"cache_read":0.075,"cache_write":0.038}},"gemini-2.5-pro":{"id":"gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":1.495,"output":9.964,"cache_read":0.242,"cache_write":0.434}},"gpt-5.6-terra":{"id":"gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.2,"output":13.199,"cache_read":0.219,"cache_write":2.749}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1.114,"output":3.899,"cache_read":0.279}},"glm-5v-turbo":{"id":"glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":202752},"cost":{"input":1.186,"output":3.955,"cache_read":0.296,"cache_write":1.544}},"mistral-medium-3.5":{"id":"mistral-medium-3.5","name":"mistral-medium-3.5","description":"Mistral Medium 3.5 is a frontier multimodal 128B model combining reasoning, coding, and instruction-following with strong agentic performance and efficient deployment.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-04-30","last_updated":"2026-04-30","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1.393,"output":7.13,"cache_read":0.139}},"minicpm-v-4.5":{"id":"minicpm-v-4.5","name":"minicpm-v-4.5","description":"MiniCPM-V 4.5 is a compact, high-performance vision-language model excelling in video understanding, OCR, and multimodal reasoning with efficient deployment.","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.651,"output":1.097}},"pixtral-12b-2409":{"id":"pixtral-12b-2409","name":"pixtral-12b-2409","description":"Pixtral 2409 12B is a state-of-the-art multimodal model with 12B parameters and a 400M vision encoder, natively trained on interleaved text and image data. It excels in tasks spanning vision-language reasoning, instruction following, and pure text understanding, making it highly effective for real-world multimodal applications.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2024-11-09","last_updated":"2024-11-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":0.223,"output":0.223}},"gpt-5":{"id":"gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.375,"output":10.96,"cache_read":0.156}},"gemini-2.5-flash":{"id":"gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65535},"cost":{"input":0.299,"output":2.491,"cache_read":0.029,"cache_write":0.097}},"claude-sonnet-4":{"id":"claude-sonnet-4","name":"Claude Sonnet 4 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":65000},"cost":{"input":2.898,"output":14.493,"cache_read":0.29,"cache_write":3.624}},"voxtral-small-2507":{"id":"voxtral-small-2507","name":"voxtral-small-2507","description":"Voxtral Small is a multimodal model with audio input, combining advanced speech capabilities with strong text performance for transcription, translation, and audio understanding.","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-02-02","last_updated":"2026-02-02","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":false,"limit":{"context":32000,"output":32000},"cost":{"input":0.111,"output":0.334,"cache_read":0.011}},"claude-sonnet-5":{"id":"claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.2,"output":11,"cache_read":0.219,"cache_write":2.749}},"qwen3-vl-235b-a22b":{"id":"qwen3-vl-235b-a22b","name":"qwen3-vl-235b-a22b","description":"Qwen3 VL 235B A22B is a 235B-parameter MoE vision-language flagship model (≈22B active) designed for frontier-level multimodal understanding across text, images, documents, and long videos.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-01-13","last_updated":"2026-01-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":0.617,"output":3.119,"cache_read":0.052}},"llama-3.3-70b-instruct":{"id":"llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131000,"output":131000},"cost":{"input":0.129,"output":0.399}},"nova-lite-v1":{"id":"nova-lite-v1","name":"nova-lite-v1","description":"Nova Lite is a fast, low-cost multimodal foundation model capable of reasoning over text, images, and video in 200+ languages.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":10000},"cost":{"input":0.069,"output":0.275}},"glm-4.7-flash":{"id":"glm-4.7-flash","name":"GLM-4.7-Flash","description":"Efficient GLM model for fast reasoning, coding, and agent workflows","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":203000,"output":203000},"cost":{"input":0.08,"output":0.478}},"nemotron-nano-v2-12b":{"id":"nemotron-nano-v2-12b","name":"nemotron-nano-v2-12b","description":"NVIDIA Nemotron Nano v2 12B is a 12-billion-parameter multimodal reasoning model designed for advanced video understanding, document intelligence, and visual reasoning, built with a hybrid Transformer-Mamba architecture for high efficiency and low latency.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2025-10-31","last_updated":"2025-10-31","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":128000},"cost":{"input":0.24,"output":0.707}}}},"hpc-ai":{"id":"hpc-ai","env":["HPC_AI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.hpc-ai.com/inference/v1","name":"HPC-AI","doc":"https://www.hpc-ai.com/doc/docs/quickstart/","models":{"minimax/minimax-m2.5":{"id":"minimax/minimax-m2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196000,"output":195000},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"anthropic/claude-opus-4.7":{"id":"anthropic/claude-opus-4.7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5}},"zai-org/glm-5.2":{"id":"zai-org/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai-org/glm-5.1":{"id":"zai-org/glm-5.1","name":"GLM 5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-08","last_updated":"2026-06-01","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202000,"output":202000},"cost":{"input":0.615,"output":2.46,"cache_read":0.133}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1002000,"output":128000},"cost":{"input":1.74,"output":3.48,"cache_read":0.145}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"moonshotai/kimi-k2.7-code":{"id":"moonshotai/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshotai/kimi-k2.5":{"id":"moonshotai/kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":256000},"cost":{"input":0.6,"output":3,"cache_read":0.1}}}},"tencent-coding-plan":{"id":"tencent-coding-plan","env":["TENCENT_CODING_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lkeap.cloud.tencent.com/coding/v3","name":"Tencent Coding Plan (China)","doc":"https://cloud.tencent.com/document/product/1772/128947","models":{"hunyuan-2.0-thinking":{"id":"hunyuan-2.0-thinking","name":"Tencent HY 2.0 Think","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-t1":{"id":"hunyuan-t1","name":"Hunyuan-T1","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-turbos":{"id":"hunyuan-turbos","name":"Hunyuan-TurboS","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"tc-code-latest":{"id":"tc-code-latest","name":"Auto","description":"Automatic model router for matching prompts to suitable backends and budgets","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi-K2.5","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-01","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hunyuan-2.0-instruct":{"id":"hunyuan-2.0-instruct","name":"Tencent HY 2.0 Instruct","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"hunyuan","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2026-03-08","last_updated":"2026-03-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"v0":{"id":"v0","env":["V0_API_KEY"],"npm":"@ai-sdk/vercel","name":"v0","doc":"https://sdk.vercel.ai/providers/ai-sdk-providers/vercel","models":{"v0-1.5-lg":{"id":"v0-1.5-lg","name":"v0-1.5-lg","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-09","last_updated":"2025-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":32000},"cost":{"input":15,"output":75}},"v0-1.5-md":{"id":"v0-1.5-md","name":"v0-1.5-md","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-06-09","last_updated":"2025-06-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":3,"output":15}},"v0-1.0-md":{"id":"v0-1.0-md","name":"v0-1.0-md","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"v0","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32000},"cost":{"input":3,"output":15}}}},"nan":{"id":"nan","env":["NAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.nan.builders/v1","name":"NaN","doc":"https://nan.builders/docs/models","models":{"qwen3.6":{"id":"qwen3.6","name":"Qwen3.6 35B-A3B","description":"Open multimodal Qwen MoE for local agents that need vision, audio, and code","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0}},"glm5.2":{"id":"glm5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents, served at 500K context on the NaN premium tier","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":500000,"output":131072},"cost":{"input":0,"output":0}},"gemma4":{"id":"gemma4","name":"Gemma 4 26B A4B IT","description":"Open Gemma instruction model for efficient chat and self-hosted deployments","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0}},"glm5.3-flash":{"id":"glm5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo-V2.5","description":"Open MiMo model for multimodal coding agents and long-context automation","family":"mimo","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0}}}},"perplexity":{"id":"perplexity","env":["PERPLEXITY_API_KEY"],"npm":"@ai-sdk/perplexity","name":"Perplexity","doc":"https://docs.perplexity.ai","models":{"sonar":{"id":"sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":1,"output":1}},"sonar-reasoning-pro":{"id":"sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"sonar-pro":{"id":"sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"sonar-deep-research":{"id":"sonar-deep-research","name":"Perplexity Sonar Deep Research","description":"Sonar search model for current answers, retrieval, and citation-backed chat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}}}},"kimi-for-coding":{"id":"kimi-for-coding","env":["KIMI_API_KEY"],"npm":"@ai-sdk/anthropic","api":"https://api.kimi.com/coding/v1","name":"Kimi For Coding","doc":"https://www.kimi.com/code/docs/en/kimi-code/models.html","models":{"kimi-for-coding-highspeed":{"id":"kimi-for-coding-highspeed","name":"Kimi For Coding HighSpeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3-256k":{"id":"k3-256k","name":"Kimi K3-256K","description":"256K-context version of Kimi K3, reducing token consumption for shorter coding sessions","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-for-coding":{"id":"kimi-for-coding","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"k3":{"id":"k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"alibaba-token-plan-cn":{"id":"alibaba-token-plan-cn","env":["ALIBABA_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1","name":"Alibaba Token Plan (China)","doc":"https://www.alibabacloud.com/help/zh/model-studio/token-plan-overview","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-r2v":{"id":"happyhorse-1.1-r2v","name":"HappyHorse 1.1 Reference-to-Video","description":"Video model for reference-guided video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"deepseek-v4-pro-0813":{"id":"deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash-0731":{"id":"deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max-preview":{"id":"qwen3.8-max-preview","name":"Qwen3.8 Max Preview","description":"Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-07-19","last_updated":"2026-07-19","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Earlier Qwen multimodal workhorse for million-token agent and document tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image-pro":{"id":"wan2.7-image-pro","name":"Wan2.7 Image Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-t2v":{"id":"happyhorse-1.1-t2v","name":"HappyHorse 1.1 Text-to-Video","description":"Video model for prompt-driven text-to-video generation","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen-image-2.0-pro":{"id":"qwen-image-2.0-pro","name":"Qwen Image 2.0 Pro","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"deepseek-v3.2":{"id":"deepseek-v3.2","name":"DeepSeek V3.2","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-03","last_updated":"2025-12-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":65536},"cost":{"input":0,"output":0}},"MiniMax-M2.5":{"id":"MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":196608,"input":196601,"output":32768},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"happyhorse-1.1-i2v":{"id":"happyhorse-1.1-i2v","name":"HappyHorse 1.1 Image-to-Video","description":"Video model for image-to-video generation","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-07-17","last_updated":"2026-07-17","modalities":{"input":["image","text"],"output":["video"]},"open_weights":false,"limit":{"context":0,"output":0},"cost":{"input":0,"output":0}},"qwen-image-2.0":{"id":"qwen-image-2.0","name":"Qwen Image 2.0","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}},"qwen3.6-flash":{"id":"qwen3.6-flash","name":"Qwen3.6 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":131072}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-27","last_updated":"2026-04-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":16384},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","min":0,"max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":98304},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal Qwen workhorse for long-context agents, visual inputs, and coding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"wan2.7-image":{"id":"wan2.7-image","name":"Wan2.7 Image","description":"Image model for prompt-driven generation, editing, and visual design workflows","attachment":false,"reasoning":false,"tool_call":false,"temperature":true,"release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text"],"output":["image"]},"open_weights":false,"limit":{"context":8192,"output":0},"cost":{"input":0,"output":0}}}},"drun":{"id":"drun","env":["DRUN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://chat.d.run/v1","name":"D.Run (China)","doc":"https://www.d.run","models":{"public/deepseek-v3":{"id":"public/deepseek-v3","name":"DeepSeek V3","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-07","release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.28,"output":1.1}},"public/minimax-m25":{"id":"public/minimax-m25","name":"MiniMax M2.5","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_details"},"temperature":true,"release_date":"2025-03-01","last_updated":"2025-03-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":204800,"output":131072},"cost":{"input":0.29,"output":1.16}},"public/deepseek-r1":{"id":"public/deepseek-r1","name":"DeepSeek R1","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2025-01-20","last_updated":"2025-01-20","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32000},"cost":{"input":0.55,"output":2.2}}}},"google-vertex-anthropic":{"id":"google-vertex-anthropic","env":["GOOGLE_VERTEX_PROJECT","GOOGLE_VERTEX_LOCATION","GOOGLE_APPLICATION_CREDENTIALS"],"npm":"@ai-sdk/google-vertex/anthropic","name":"Vertex (Anthropic)","doc":"https://cloud.google.com/vertex-ai/generative-ai/docs/partner-models/claude","models":{"claude-sonnet-4@20250514":{"id":"claude-sonnet-4@20250514","name":"Claude Sonnet 4","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"status":"deprecated","cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"claude-opus-4-5@20251101":{"id":"claude-opus-4-5@20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-sonnet-4-6@default":{"id":"claude-sonnet-4-6@default","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75,"tiers":[{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":6,"output":22.5,"cache_read":0.6,"cache_write":7.5}}},"claude-fable-5@default":{"id":"claude-fable-5@default","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"claude-opus-4-6@default":{"id":"claude-opus-4-6@default","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]},{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-4@20250514":{"id":"claude-opus-4@20250514","name":"Claude Opus 4","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-05-22","last_updated":"2025-05-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-haiku-4-5@20251001":{"id":"claude-haiku-4-5@20251001","name":"Claude Haiku 4.5","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":1,"output":5,"cache_read":0.1,"cache_write":1.25}},"claude-sonnet-5@default":{"id":"claude-sonnet-5@default","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"claude-opus-4-1@20250805":{"id":"claude-opus-4-1@20250805","name":"Claude Opus 4.1","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":32000},"status":"deprecated","cost":{"input":15,"output":75,"cache_read":1.5,"cache_write":18.75}},"claude-fable-5-1@default":{"id":"claude-fable-5-1@default","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"claude-opus-4-7@default":{"id":"claude-opus-4-7@default","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-opus-5@default":{"id":"claude-opus-5@default","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"claude-opus-4-8@default":{"id":"claude-opus-4-8@default","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25,"tiers":[{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":10,"output":37.5,"cache_read":1,"cache_write":12.5}}},"claude-sonnet-4-5@20250929":{"id":"claude-sonnet-4-5@20250929","name":"Claude Sonnet 4.5","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"budget_tokens","min":1024}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}}}},"anyapi":{"id":"anyapi","env":["ANYAPI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.anyapi.ai/v1","name":"AnyAPI","doc":"https://docs.anyapi.ai","models":{"mistralai/devstral-2512":{"id":"mistralai/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"status":"deprecated"},"mistralai/mistral-large-2512":{"id":"mistralai/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":127999}],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":30,"output":150,"cache_read":3,"cache_write":37.5},"provider":{"body":{"speed":"fast"},"headers":{"anthropic-beta":"fast-mode-2026-02-01"}}}}}},"anthropic/claude-haiku-4-5":{"id":"anthropic/claude-haiku-4-5","name":"Claude Haiku 4.5 (latest)","description":"Fast Claude lane for lightweight agents, office tasks, and responsive chat","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2025-10-15","last_updated":"2025-10-15","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"anthropic/claude-sonnet-4-5":{"id":"anthropic/claude-sonnet-4-5","name":"Claude Sonnet 4.5 (latest)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high"]},{"type":"budget_tokens","min":1024,"max":63999}],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2025-09-29","last_updated":"2025-09-29","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000}},"google/gemini-2.5-flash-lite":{"id":"google/gemini-2.5-flash-lite","name":"Gemini 2.5 Flash-Lite","description":"Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-3-pro-preview":{"id":"google/gemini-3-pro-preview","name":"Gemini 3 Pro Preview","description":"Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-18","last_updated":"2025-11-18","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-pro":{"id":"google/gemini-2.5-pro","name":"Gemini 2.5 Pro","description":"Google's proven reasoning model for coding, math, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"google/gemini-2.5-flash":{"id":"google/gemini-2.5-flash","name":"Gemini 2.5 Flash","description":"Fast Gemini workhorse for multimodal apps where latency and price matter","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-06-17","last_updated":"2025-06-17","modalities":{"input":["text","image","audio","video","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-r1":{"id":"deepseek/deepseek-r1","name":"DeepSeek Reasoner","description":"DeepSeek reasoning model for multi-step analysis, math, coding, and tools","family":"deepseek-thinking","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"experimental":{"modes":{"fast":{"cost":{"input":5,"output":30,"cache_read":0.5},"provider":{"body":{"service_tier":"priority"}}}}}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000}},"perplexity/sonar-reasoning-pro":{"id":"perplexity/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096}},"perplexity/sonar-pro":{"id":"perplexity/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192}}}},"opencode-go":{"id":"opencode-go","env":["OPENCODE_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://opencode.ai/zen/go/v1","name":"OpenCode Go","doc":"https://opencode.ai/docs/zen","models":{"qwen3.7-max":{"id":"qwen3.7-max","name":"Qwen3.7 Max","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"qwen3.7-max","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-05-21","last_updated":"2026-05-21","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":2.5,"output":7.5,"cache_read":0.5,"cache_write":3.125}},"longcat-2.0":{"id":"longcat-2.0","name":"LongCat-2.0","description":"Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window","family":"longcat","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepseek-v4-flash-vision-exp":{"id":"deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"qwen3.6-plus":{"id":"qwen3.6-plus","name":"Qwen3.6 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.6","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.5,"output":3,"cache_read":0.05,"cache_write":0.625,"tiers":[{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.2,"cache_write":2.5}}},"muse-spark-1.2-contributor":{"id":"muse-spark-1.2-contributor","name":"Muse Spark 1.2 Contributor","description":"Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-05","last_updated":"2026-08-05","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"minimax-m2.7":{"id":"minimax-m2.7","name":"MiniMax-M2.7","description":"MiniMax model for chat, coding, office work, and agentic tasks","family":"minimax-m2.7","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"kimi-k2.6":{"id":"kimi-k2.6","name":"Kimi K2.6","description":"Kimi multimodal agent model for visual understanding, coding, and planning","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"glm-5.2":{"id":"glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"minimax-m2.5":{"id":"minimax-m2.5","name":"MiniMax-M2.5","description":"Legacy model retained for compatibility with older integrations","family":"minimax-m2.5","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":65536},"status":"deprecated","provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax-m3":{"id":"minimax-m3","name":"MiniMax-M3","description":"MiniMax multimodal coding model for long-context reasoning and agent tasks","family":"minimax-m3","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"}],"tool_call":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-31","last_updated":"2026-05-31","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.3,"output":1.2,"cache_read":0.06,"tiers":[{"input":0.6,"output":2.4,"cache_read":0.12,"tier":{"type":"context","size":512000}}],"context_over_200k":{"input":0.6,"output":2.4,"cache_read":0.12}}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"kimi-k2.7-code":{"id":"kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"grok-4.5":{"id":"grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"status":"deprecated","provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"ox-alpha-free":{"id":"ox-alpha-free","name":"Ox Alpha Free (Unlimited)","description":"Stealth reasoning model for coding, agentic tasks, and tool use","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"status":"deprecated","cost":{"input":0,"output":0,"cache_read":0}},"deepseek-v4.1-flash":{"id":"deepseek-v4.1-flash","name":"DeepSeek V4.1 Flash","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.15,"output":0.6,"cache_read":0.003}},"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}},"omen-alpha":{"id":"omen-alpha","name":"Omen Alpha","description":"oH man anothEr aLPha ModEl","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":128000},"status":"deprecated","cost":{"input":0.2,"output":0.66,"cache_read":0.04}},"gpt-5.6-luna":{"id":"gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"glm-5.3-flash":{"id":"glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"muse-spark-1.3-contributor":{"id":"muse-spark-1.3-contributor","name":"Muse Spark 1.3 Contributor","description":"Muse Spark 1.3 is a multimodal reasoning model from Meta for coding and agentic workflows.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","pdf","audio"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":131072},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":0.1,"output":0.2,"cache_read":0.002}},"mimo-v2-pro":{"id":"mimo-v2-pro","name":"MiMo V2 Pro","description":"Legacy model retained for compatibility with older integrations","family":"mimo-v2-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"status":"deprecated","cost":{"input":1,"output":3,"cache_read":0.2,"tiers":[{"input":2,"output":6,"cache_read":0.4,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":2,"output":6,"cache_read":0.4}}},"qwen3.8-flash":{"id":"qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens"}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"provider":{"npm":"@ai-sdk/anthropic"},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"grok-4.6":{"id":"grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"provider":{"npm":"@ai-sdk/openai"},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"glm-5":{"id":"glm-5","name":"GLM-5","description":"Legacy model retained for compatibility with older integrations","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-02-11","last_updated":"2026-02-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"status":"deprecated","cost":{"input":1,"output":3.2,"cache_read":0.2}},"mimo-v2-omni":{"id":"mimo-v2-omni","name":"MiMo V2 Omni","description":"Legacy model retained for compatibility with older integrations","family":"mimo-v2-omni","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text","image","audio","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"status":"deprecated","cost":{"input":0.4,"output":2,"cache_read":0.08}},"qwen3.8-max":{"id":"qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter multimodal flagship for coding, professional work, and long-horizon agentic workflows","family":"qwen3.8-max","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","xhigh"]},{"type":"budget_tokens","max":262144}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"kimi-k2.5":{"id":"kimi-k2.5","name":"Kimi K2.5","description":"Legacy model retained for compatibility with older integrations","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-10","release_date":"2026-01-27","last_updated":"2026-01-27","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.6,"output":3,"cache_read":0.1}},"glm-5.1":{"id":"glm-5.1","name":"GLM-5.1","description":"Flagship GLM model for hybrid reasoning, coding, and agentic engineering","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2025-04","release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":32768},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"qwen3.7-plus":{"id":"qwen3.7-plus","name":"Qwen3.7 Plus","description":"Multimodal reasoning model for visual analysis, planning, and tool use","family":"qwen3.7-plus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":262144}],"tool_call":true,"temperature":true,"release_date":"2026-06-02","last_updated":"2026-06-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.4,"output":1.6,"cache_read":0.04,"cache_write":0.5,"tiers":[{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5,"tier":{"type":"context","size":256000}}],"context_over_200k":{"input":1.2,"output":4.8,"cache_read":0.12,"cache_write":1.5}}},"deepseek-v4-pro":{"id":"deepseek-v4-pro","name":"DeepSeek V4 Pro (New)","description":"Flagship DeepSeek model for coding, reasoning, and agentic work","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.66,"output":1.98,"cache_read":0.022}},"glm-5.3":{"id":"glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"interleaved":{"field":"reasoning_content"},"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"mimo-v2.5":{"id":"mimo-v2.5","name":"MiMo V2.5","description":"MiMo omni model for text, image, video, audio, and agents","family":"mimo-v2.5","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text","image","audio","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":128000},"cost":{"input":0.14,"output":0.28,"cache_read":0.0028}},"mimo-v2.5-pro":{"id":"mimo-v2.5-pro","name":"MiMo V2.5 Pro","description":"MiMo pro model for strong multimodal reasoning and agent execution","family":"mimo-v2.5-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"interleaved":{"field":"reasoning_content"},"temperature":true,"knowledge":"2024-12","release_date":"2026-04-22","last_updated":"2026-04-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":0.435,"output":0.87,"cache_read":0.003625}},"qwen3.5-plus":{"id":"qwen3.5-plus","name":"Qwen3.5 Plus","description":"Legacy model retained for compatibility with older integrations","family":"qwen3.5","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"budget_tokens","max":81920}],"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-02-16","last_updated":"2026-02-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"status":"deprecated","cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25}}}},"tencent-token-plan":{"id":"tencent-token-plan","env":["TENCENT_TOKEN_PLAN_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.lkeap.cloud.tencent.com/plan/v3","name":"Tencent Token Plan","doc":"https://cloud.tencent.com/document/product/1823/130060","models":{"hy3":{"id":"hy3","name":"Hy3","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"input":192000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"hy4-preview":{"id":"hy4-preview","name":"Hy4 preview","description":"A next-generation productivity model with significantly enhanced Agent and complex task execution capabilities.","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["none","high"]}],"tool_call":true,"temperature":true,"release_date":"2026-08-28","last_updated":"2026-08-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":64000},"cost":{"input":0.834,"output":2.501,"cache_read":0.042}}}},"gitlab":{"id":"gitlab","env":["GITLAB_TOKEN"],"npm":"gitlab-ai-provider","name":"GitLab Duo","doc":"https://docs.gitlab.com/user/duo_agent_platform/","models":{"duo-chat-gpt-5-6-luna":{"id":"duo-chat-gpt-5-6-luna","name":"Agentic Chat (GPT-5.6 Luna)","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-5":{"id":"duo-chat-opus-5","name":"Agentic Chat (Claude Opus 5)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-8":{"id":"duo-chat-opus-4-8","name":"Agentic Chat (Claude Opus 4.8)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-1":{"id":"duo-chat-gpt-5-1","name":"Agentic Chat (GPT-5.1)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-2":{"id":"duo-chat-gpt-5-2","name":"Agentic Chat (GPT-5.2)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-23","last_updated":"2026-01-23","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-4-nano":{"id":"duo-chat-gpt-5-4-nano","name":"Agentic Chat (GPT-5.4 Nano)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-haiku-4-5":{"id":"duo-chat-haiku-4-5","name":"Agentic Chat (Claude Haiku 4.5)","description":"Fast Claude model for responsive assistance, classification, and lightweight agents","family":"claude-haiku","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-02-28","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-6":{"id":"duo-chat-opus-4-6","name":"Agentic Chat (Claude Opus 4.6)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-6-terra":{"id":"duo-chat-gpt-5-6-terra","name":"Agentic Chat (GPT-5.6 Terra)","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-sonnet-5":{"id":"duo-chat-sonnet-5","name":"Agentic Chat (Claude Sonnet 5)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"toggle"},{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-5":{"id":"duo-chat-opus-4-5","name":"Agentic Chat (Claude Opus 4.5)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-6-astra":{"id":"duo-chat-gpt-6-astra","name":"Agentic Chat (GPT-6 Astra)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-3-codex":{"id":"duo-chat-gpt-5-3-codex","name":"Agentic Chat (GPT-5.3 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-fable-5-1":{"id":"duo-chat-fable-5-1","name":"Agentic Chat (Claude Fable 5.1)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-4-mini":{"id":"duo-chat-gpt-5-4-mini","name":"Agentic Chat (GPT-5.4 Mini)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-sonnet-4-6":{"id":"duo-chat-sonnet-4-6","name":"Agentic Chat (Claude Sonnet 4.6)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-02-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-5":{"id":"duo-chat-gpt-5-5","name":"Agentic Chat (GPT-5.5)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-fable-5":{"id":"duo-chat-fable-5","name":"Agentic Chat (Claude Fable 5)","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-opus-4-7":{"id":"duo-chat-opus-4-7","name":"Agentic Chat (Claude Opus 4.7)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-4":{"id":"duo-chat-gpt-5-4","name":"Agentic Chat (GPT-5.4)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-6-sol":{"id":"duo-chat-gpt-5-6-sol","name":"Agentic Chat (GPT-5.6 Sol)","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}},"duo-chat-gpt-5-2-codex":{"id":"duo-chat-gpt-5-2-codex","name":"Agentic Chat (GPT-5.2 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-codex":{"id":"duo-chat-gpt-5-codex","name":"Agentic Chat (GPT-5 Codex)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-gpt-5-mini":{"id":"duo-chat-gpt-5-mini","name":"Agentic Chat (GPT-5 Mini)","description":"Chat-tuned GPT model for conversational assistance, writing, and tool workflows","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2026-01-22","last_updated":"2026-01-22","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0,"output":0}},"duo-chat-sonnet-4-5":{"id":"duo-chat-sonnet-4-5","name":"Agentic Chat (Claude Sonnet 4.5)","description":"Balanced Claude model for coding, analysis, agent workflows, and cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"temperature":true,"knowledge":"2025-07-31","release_date":"2026-01-08","last_updated":"2026-01-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":0,"output":0,"cache_read":0,"cache_write":0}}}},"neosmith":{"id":"neosmith","env":["NEOSMITH_API_KEY"],"npm":"@ai-sdk/openai","api":"https://router.neosmith.ai/v1","name":"NeoSmith","doc":"https://neosmith.ai/docs","models":{"neosmith.intelligent-maestro":{"id":"neosmith.intelligent-maestro","name":"NeoSmith Maestro","description":"Highest-accuracy coding tier. Hard, self-contained problems run NeoSmith's premium multi-model solver; everything else gets the strongest intelligence tier.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-08","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2.4,"output":12,"cache_read":0.35,"cache_write":0}},"neosmith.intelligent-basic":{"id":"neosmith.intelligent-basic","name":"NeoSmith Basic","description":"Cost-capped tier. Intelligent routing with a Claude Sonnet ceiling — Opus is never invoked.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.17,"output":4.37,"cache_read":0.22,"cache_write":0}},"neosmith.intelligent-pro":{"id":"neosmith.intelligent-pro","name":"NeoSmith Pro","description":"Default production tier. Intelligent NeoSmith routing with a Claude Opus ceiling on escalation.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-04-26","last_updated":"2026-07-18","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":1.81,"output":8.39,"cache_read":0.3,"cache_write":0}},"neosmith.neolite":{"id":"neosmith.neolite","name":"NeoSmith NeoLite","description":"Sealed single-model budget tier. 512K context, text and images, tool use, and no escalation of any kind.","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-06-20","last_updated":"2026-08-03","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":512000,"output":64000},"cost":{"input":0.6,"output":2.4,"cache_read":0.08,"cache_write":0}}}},"tinfoil":{"id":"tinfoil","env":["TINFOIL_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://inference.tinfoil.sh/v1","name":"Tinfoil","doc":"https://docs.tinfoil.sh","models":{"nomic-embed-text":{"id":"nomic-embed-text","name":"Nomic Embed Text v1.5","description":"Embedding model for semantic search, retrieval, clustering, and ranking pipelines","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":false,"release_date":"2024-02","last_updated":"2024-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":8192,"output":768},"cost":{"input":0.05,"output":0}},"gemma4-31b":{"id":"gemma4-31b","name":"Gemma 4 31B IT","description":"Largest Gemma 4 instruction model for open, self-hosted chat and reasoning","family":"gemma","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-02","last_updated":"2026-04-02","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":32768},"cost":{"input":0.4,"output":1}},"deepseek-v4-flash":{"id":"deepseek-v4-flash","name":"DeepSeek V4 Flash 0731","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":0.7,"cache_read":0.06}},"gpt-oss-safeguard-120b":{"id":"gpt-oss-safeguard-120b","name":"gpt-oss-safeguard-120b","description":"Safety model for policy screening, moderation, and risk-aware routing workflows","family":"gpt-oss","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-10-29","last_updated":"2025-10-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"kimi-k3":{"id":"kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":131072},"cost":{"input":4,"output":20,"cache_read":0.8}},"glm-5-3-flash":{"id":"glm-5-3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.4,"output":1.25,"cache_read":0.1}},"gpt-oss-120b":{"id":"gpt-oss-120b","name":"gpt-oss-120b","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"llama3-3-70b":{"id":"llama3-3-70b","name":"Llama-3.3-70B-Instruct","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":true,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":1.75,"output":2.75}}}},"edenai":{"id":"edenai","env":["EDENAI_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"https://api.edenai.run/v3","name":"Eden AI","doc":"https://docs.edenai.co","models":{"qwen/deepseek-v4-pro-0813":{"id":"qwen/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Alibaba)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.5808,"output":1.7424}},"qwen/deepseek-v4-flash-0731":{"id":"qwen/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Alibaba)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.176,"output":0.528}},"qwen/qwen3-coder-plus":{"id":"qwen/qwen3-coder-plus","name":"Qwen3 Coder Plus","description":"Hosted Qwen coder for software agents, repo edits, and long-context code","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":1,"output":5,"cache_read":0.2,"cache_write":1.25}},"qwen/qwen3-next-80b-a3b-thinking":{"id":"qwen/qwen3-next-80b-a3b-thinking","name":"Qwen3-Next 80B-A3B (Thinking)","description":"Efficient Qwen thinking model for local reasoning, math, and coding agents","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen-vl-max":{"id":"qwen/qwen-vl-max","name":"Qwen-VL Max","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-08","last_updated":"2025-08-13","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":3.2,"cache_read":0.16}},"qwen/qwen3-next-80b-a3b-instruct":{"id":"qwen/qwen3-next-80b-a3b-instruct","name":"Qwen3-Next 80B-A3B Instruct","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09","last_updated":"2025-09","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":1.2}},"qwen/qwen3-coder-flash":{"id":"qwen/qwen3-coder-flash","name":"Qwen3 Coder Flash","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-28","last_updated":"2025-07-28","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":65536},"cost":{"input":0.3,"output":1.5,"cache_read":0.06,"cache_write":0.375}},"qwen/qwen-max":{"id":"qwen/qwen-max","name":"Qwen Max","description":"Flagship Qwen model for complex reasoning, coding, and agentic workflows","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2024-04-03","last_updated":"2025-01-25","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":32768,"output":8192},"cost":{"input":1.6,"output":6.4,"cache_read":0.32}},"qwen/qwen-vl-plus":{"id":"qwen/qwen-vl-plus","name":"Qwen-VL Plus","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2024-01-25","last_updated":"2025-08-15","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.21,"output":0.63,"cache_read":0.042}},"qwen/qwen3.8-27b":{"id":"qwen/qwen3.8-27b","name":"Qwen3.8 27B","description":"Dense 27B vision-language model for coding, agent tasks, and image and video understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":32768},"cost":{"input":0.5,"output":3,"cache_read":0.1,"cache_write":0.625}},"qwen/qwen3-max@eu":{"id":"qwen/qwen3-max@eu","name":"Qwen3 Max (EU)","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"qwen/qwq-plus":{"id":"qwen/qwq-plus","name":"QwQ Plus","description":"Qwen reasoning model for deliberate problem solving, math, and coding","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-04","release_date":"2025-03-05","last_updated":"2025-03-05","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":8192},"cost":{"input":0.8,"output":2.4}},"qwen/qwen3-coder-next":{"id":"qwen/qwen3-coder-next","name":"Qwen3 Coder Next","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-coder-30b-a3b-instruct":{"id":"qwen/qwen3-coder-30b-a3b-instruct","name":"Qwen3-Coder 30B-A3B Instruct","description":"Smaller Qwen coder for efficient local agents and repo-level fixes","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.45,"output":2.25}},"qwen/qwen3.8-max-0902":{"id":"qwen/qwen3.8-max-0902","name":"Qwen3.8 Max 0902","description":"2026-09-02 upgraded snapshot of Qwen3.8 Max with stronger coding, collaborative agents, and multimodal document understanding","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-max":{"id":"qwen/qwen3-max","name":"Qwen3 Max","description":"Flagship Qwen3 model for coding agents, complex reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":65536},"cost":{"input":1.2,"output":6,"cache_read":0.24,"cache_write":1.5}},"qwen/qwen3.8-2.4t-a95b":{"id":"qwen/qwen3.8-2.4t-a95b","name":"Qwen3.8 2.4T A95B","description":"Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows","family":"qwen","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3.8-flash":{"id":"qwen/qwen3.8-flash","name":"Qwen3.8 Flash","description":"Qwen vision-language model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":0.15,"output":0.47,"cache_read":0.016,"cache_write":0.2}},"qwen/qwen3.8-max":{"id":"qwen/qwen3.8-max","name":"Qwen3.8 Max","description":"2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-03","last_updated":"2026-08-03","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":131072},"cost":{"input":2,"output":6,"cache_read":0.25,"cache_write":2.5}},"qwen/qwen3-coder-next@eu":{"id":"qwen/qwen3-coder-next@eu","name":"Qwen3 Coder Next (EU)","description":"Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2026-02-03","last_updated":"2026-02-03","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0.3,"output":1.5}},"qwen/qwen3-vl-235b-a22b-thinking":{"id":"qwen/qwen3-vl-235b-a22b-thinking","name":"Qwen3 VL 235B A22B Thinking","description":"Qwen vision-language thinking model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":4}},"qwen/qwen3-vl-235b-a22b-instruct":{"id":"qwen/qwen3-vl-235b-a22b-instruct","name":"Qwen3 VL 235B A22B Instruct","description":"Qwen vision-language instruct model for visual reasoning, documents, and agent tasks","family":"qwen","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-03-31","release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.4,"output":1.6}},"qwen/qwen3-235b-a22b-instruct-2507":{"id":"qwen/qwen3-235b-a22b-instruct-2507","name":"Qwen3 235B-A22B Instruct 2507","description":"Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-07-21","last_updated":"2025-07-21","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":16384},"cost":{"input":0.23,"output":0.92}},"qwen/qwen3-coder-480b-a35b-instruct":{"id":"qwen/qwen3-coder-480b-a35b-instruct","name":"Qwen3-Coder 480B-A35B Instruct","description":"Open Qwen coding heavyweight for repository reasoning and agentic engineering","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-04","last_updated":"2025-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":1.5,"output":7.5}},"groq/openai/gpt-oss-20b":{"id":"groq/openai/gpt-oss-20b","name":"GPT OSS 20B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.075,"output":0.3,"cache_read":0.0375}},"groq/openai/gpt-oss-120b":{"id":"groq/openai/gpt-oss-120b","name":"GPT OSS 120B (Groq)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"scaleway/deepseek-v4-flash-0731":{"id":"scaleway/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Scaleway)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":384000},"cost":{"input":0.46464,"output":0.92928}},"scaleway/gpt-oss-120b":{"id":"scaleway/gpt-oss-120b","name":"GPT OSS 120B (Scaleway)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.17424,"output":0.69696}},"scaleway/llama-3.3-70b-instruct":{"id":"scaleway/llama-3.3-70b-instruct","name":"Llama-3.3-70B-Instruct (Scaleway)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":1.045439,"output":1.045439}},"nebius/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"nebius/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Nebius)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1024000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.14}},"nebius/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"nebius/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Nebius)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":979000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":1.32}},"nebius/nvidia/Nemotron-3-Ultra-550b-a55b":{"id":"nebius/nvidia/Nemotron-3-Ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B (Nebius)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":128000},"cost":{"input":1,"output":3,"cache_read":1}},"nebius/nvidia/nemotron-3-super-120b-a12b":{"id":"nebius/nvidia/nemotron-3-super-120b-a12b","name":"Nemotron 3 Super 120B A12B (Nebius)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.3,"output":0.9,"cache_read":0.3}},"nebius/meta-llama/Llama-3.3-70B-Instruct":{"id":"nebius/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (Nebius)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.13,"output":0.4,"cache_read":0.13}},"nebius/openai/gpt-oss-120b":{"id":"nebius/openai/gpt-oss-120b","name":"GPT OSS 120B (Nebius)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.15}},"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct":{"id":"cloudflare/@cf/qwen/qwen2.5-coder-32b-instruct","name":"Qwen2.5-Coder-32B-Instruct (Cloudflare)","description":"Open coding-focused Qwen model for code generation, repair, and repository reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2024-11-12","last_updated":"2024-11-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":8192},"cost":{"input":0.66,"output":1}},"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813":{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Cloudflare)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731":{"id":"cloudflare/@cf/deepseek-ai/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Cloudflare)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1310720,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"cloudflare/@cf/zai-org/glm-4.7-flash":{"id":"cloudflare/@cf/zai-org/glm-4.7-flash","name":"GLM-4.7-Flash (Cloudflare)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.0605,"output":0.4}},"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it":{"id":"cloudflare/@cf/aisingapore/gemma-sea-lion-v4-27b-it","name":"Gemma-SEA-LION-v4-27B-IT (Cloudflare)","description":"Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following","family":"gemma","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-09-23","last_updated":"2025-09-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":128000},"cost":{"input":0.351,"output":0.555}},"cloudflare/@cf/meta/llama-guard-3-8b":{"id":"cloudflare/@cf/meta/llama-guard-3-8b","name":"Llama-Guard-3-8B (Cloudflare)","description":"Llama 3.1-based safety classifier for moderating prompts and model responses","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.484,"output":0.03}},"cloudflare/@cf/openai/gpt-oss-20b":{"id":"cloudflare/@cf/openai/gpt-oss-20b","name":"GPT OSS 20B (Cloudflare)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.2,"output":0.3}},"cloudflare/@cf/openai/gpt-oss-120b":{"id":"cloudflare/@cf/openai/gpt-oss-120b","name":"GPT OSS 120B (Cloudflare)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":32768},"cost":{"input":0.35,"output":0.75}},"minimax/MiniMax-M2":{"id":"minimax/MiniMax-M2","name":"MiniMax-M2","description":"Efficient open MiniMax model built for coding agents and tool-heavy workflows","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-10-27","last_updated":"2025-10-27","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2}},"minimax/MiniMax-M2.1":{"id":"minimax/MiniMax-M2.1","name":"MiniMax-M2.1","description":"Earlier MiniMax agent model for practical coding and productivity tasks","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-23","last_updated":"2025-12-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/MiniMax-M2.5":{"id":"minimax/MiniMax-M2.5","name":"MiniMax-M2.5","description":"Prior MiniMax coding model for agent workflows, office edits, and automation","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.03}},"minimax/MiniMax-M3":{"id":"minimax/MiniMax-M3","name":"MiniMax-M3","description":"MiniMax multimodal model for long-context coding, perception, and agent planning","family":"minimax","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-01","last_updated":"2026-06-01","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":512000},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"minimax/MiniMax-M2.7":{"id":"minimax/MiniMax-M2.7","name":"MiniMax-M2.7","description":"Open MiniMax flagship for coding agents, office automation, and complex environments","family":"minimax","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-18","last_updated":"2026-03-18","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":204800,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.06}},"ovhcloud/gpt-oss-20b":{"id":"ovhcloud/gpt-oss-20b","name":"GPT OSS 20B (OVHcloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.18}},"ovhcloud/gpt-oss-120b":{"id":"ovhcloud/gpt-oss-120b","name":"GPT OSS 120B (OVHcloud)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.09,"output":0.47}},"anthropic/claude-sonnet-4-6":{"id":"anthropic/claude-sonnet-4-6","name":"Claude Sonnet 4.6","description":"Claude workhorse for coding agents, careful analysis, and production cost control","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-17","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":64000},"cost":{"input":3,"output":15,"cache_read":0.3,"cache_write":3.75}},"anthropic/claude-fable-latest":{"id":"anthropic/claude-fable-latest","name":"Claude Fable Latest (Claude Fable 5.1)","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-5":{"id":"anthropic/claude-opus-5","name":"Claude Opus 5","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-5":{"id":"anthropic/claude-opus-4-5","name":"Claude Opus 4.5 (latest)","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-24","last_updated":"2025-11-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5-1":{"id":"anthropic/claude-fable-5-1","name":"Claude Fable 5.1","description":"Claude model for demanding reasoning and long-horizon agentic work","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-06","release_date":"2026-09-01","last_updated":"2026-09-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":0.25,"cache_write":12.5}},"anthropic/claude-opus-4-6":{"id":"anthropic/claude-opus-4-6","name":"Claude Opus 4.6","description":"High-end Claude for difficult coding, planning, and slower expert reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05-31","release_date":"2026-02-05","last_updated":"2026-03-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-latest":{"id":"anthropic/claude-opus-latest","name":"Claude Opus Latest (Claude Opus 5)","description":"Strongest Claude Opus model for coding, agents, and professional work","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-05","release_date":"2026-07-24","last_updated":"2026-07-24","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-opus-4-7":{"id":"anthropic/claude-opus-4-7","name":"Claude Opus 4.7","description":"Stronger Opus tier for advanced software work and high-stakes reasoning","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-fable-5":{"id":"anthropic/claude-fable-5","name":"Claude Fable 5","description":"Claude model for creative writing, analysis, and controlled agent workflows","family":"claude-fable","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-09","last_updated":"2026-06-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5}},"anthropic/claude-sonnet-latest":{"id":"anthropic/claude-sonnet-latest","name":"Claude Sonnet Latest (Claude Sonnet 5)","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-8":{"id":"anthropic/claude-opus-4-8","name":"Claude Opus 4.8","description":"Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"anthropic/claude-sonnet-5":{"id":"anthropic/claude-sonnet-5","name":"Claude Sonnet 5","description":"Everyday Claude agent model for coding, planning, browsing, and general work","family":"claude-sonnet","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-01-31","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":128000},"cost":{"input":2,"output":10,"cache_read":0.2,"cache_write":2.5}},"anthropic/claude-opus-4-5-20251101":{"id":"anthropic/claude-opus-4-5-20251101","name":"Claude Opus 4.5","description":"Flagship Claude model for deep reasoning, coding, and long-horizon agents","family":"claude-opus","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-11-01","last_updated":"2025-11-01","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":64000},"cost":{"input":5,"output":25,"cache_read":0.5,"cache_write":6.25}},"google/gemini-pro-latest":{"id":"google/gemini-pro-latest","name":"Gemini Pro Latest (Gemini 3.1 Pro Preview)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-pro-preview-customtools":{"id":"google/gemini-3.1-pro-preview-customtools","name":"Gemini 3.1 Pro Preview Custom Tools","description":"Advanced Gemini model for complex reasoning, coding, and multimodal analysis","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.1-flash-lite-image":{"id":"google/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5}},"google/gemini-2.5-flash-image":{"id":"google/gemini-2.5-flash-image","name":"Nano Banana","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["audio","image","text","video"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":1}},"google/gemini-3-pro-image":{"id":"google/gemini-3-pro-image","name":"Nano Banana Pro","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"google/gemini-3.1-pro-preview":{"id":"google/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4}}},"google/gemini-3.6-flash":{"id":"google/gemini-3.6-flash","name":"Gemini 3.6 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-flash-lite":{"id":"google/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"google/gemini-3.5-flash":{"id":"google/gemini-3.5-flash","name":"Gemini 3.5 Flash","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"google/gemini-3.1-flash-lite-preview":{"id":"google/gemini-3.1-flash-lite-preview","name":"Gemini 3.1 Flash Lite Preview","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-03-03","last_updated":"2026-03-03","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"google/gemini-3.1-flash-image":{"id":"google/gemini-3.1-flash-image","name":"Nano Banana 2","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":0.5,"output":3}},"google/gemini-3.5-flash-lite":{"id":"google/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"google/gemini-3-pro-image-preview":{"id":"google/gemini-3-pro-image-preview","name":"Nano Banana Pro Preview","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-11-20","last_updated":"2025-11-20","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"google/gemini-3-flash-preview":{"id":"google/gemini-3-flash-preview","name":"Gemini 3 Flash Preview","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333,"input_audio":1}},"google/gemini-3.8-flash":{"id":"google/gemini-3.8-flash","name":"Gemini 3.8 Flash","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.7-flash":{"id":"google/gemini-3.7-flash","name":"Gemini 3.7 Flash","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-flash-latest":{"id":"google/gemini-flash-latest","name":"Gemini Flash Latest (Gemini 3.8 Flash)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"google/gemini-3.1-flash-image-preview":{"id":"google/gemini-3.1-flash-image-preview","name":"Nano Banana 2 Preview","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-26","last_updated":"2026-02-26","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":65536},"cost":{"input":0.5,"output":3}},"tensorx/deepseek/deepseek-v4-pro-0813":{"id":"tensorx/deepseek/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (TensorX)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":2,"output":4,"cache_read":0.5}},"tensorx/deepseek/deepseek-v4-flash-0731":{"id":"tensorx/deepseek/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (TensorX)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.25,"output":0.3,"cache_read":0.0625}},"tensorx/moonshotai/kimi-k2.5":{"id":"tensorx/moonshotai/kimi-k2.5","name":"Kimi K2.5 (TensorX)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":2.8,"cache_read":0.125}},"flexai/Nemotron-3-Super-120B-A12B":{"id":"flexai/Nemotron-3-Super-120B-A12B","name":"Nemotron 3 Super 120B A12B (FlexAI)","description":"Nemotron middle tier for collaborative agents and high-volume reasoning workloads","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2026-03-11","last_updated":"2026-03-11","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.085,"output":0.4}},"flexai/Step-3.7-Flash":{"id":"flexai/Step-3.7-Flash","name":"Step 3.7 Flash (FlexAI)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15}},"flexai/gpt-oss-20b":{"id":"flexai/gpt-oss-20b","name":"GPT OSS 20B (FlexAI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.13}},"flexai/DeepSeek-V4-Flash-0731":{"id":"flexai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (FlexAI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":786432,"output":384000},"cost":{"input":0.065,"output":0.18}},"flexai/Muse-Glimmer-30B":{"id":"flexai/Muse-Glimmer-30B","name":"Muse Glimmer 30B (FlexAI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.1}},"flexai/gpt-oss-120b":{"id":"flexai/gpt-oss-120b","name":"GPT OSS 120B (FlexAI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.037,"output":0.17}},"databricks/databricks-gpt-oss-20b@eu":{"id":"databricks/databricks-gpt-oss-20b@eu","name":"GPT OSS 20B (Databricks, EU)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.30002,"cache_read":0.007,"cache_write":0.07}},"databricks/databricks-deepseek-v4-pro-0813":{"id":"databricks/databricks-deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Databricks)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.31999,"output":3.95997,"cache_read":0.13202,"cache_write":1.31999}},"databricks/databricks-gpt-oss-120b@eu":{"id":"databricks/databricks-gpt-oss-120b@eu","name":"GPT OSS 120B (Databricks, EU)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15001,"output":0.59997,"cache_read":0.015001,"cache_write":0.15001}},"databricks/databricks-gpt-oss-20b":{"id":"databricks/databricks-gpt-oss-20b","name":"GPT OSS 20B (Databricks)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.07,"output":0.30002,"cache_read":0.007,"cache_write":0.07}},"databricks/databricks-deepseek-v4-flash-0731":{"id":"databricks/databricks-deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Databricks)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.028,"cache_write":0.14}},"databricks/databricks-inkling":{"id":"databricks/databricks-inkling","name":"Inkling (Databricks)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":1048576},"cost":{"input":1.00002,"output":4.04999,"cache_read":0.17003,"cache_write":1.00002}},"databricks/databricks-gpt-oss-120b":{"id":"databricks/databricks-gpt-oss-120b","name":"GPT OSS 120B (Databricks)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15001,"output":0.59997,"cache_read":0.015001,"cache_write":0.15001}},"deepinfra/nemotron-3-ultra-550b-a55b":{"id":"deepinfra/nemotron-3-ultra-550b-a55b","name":"Nemotron 3 Ultra 550B A55B (Deep Infra)","description":"Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-04","last_updated":"2026-06-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.5,"output":2.2,"cache_read":0.1}},"deepinfra/ByteDance/Seed-2.0-mini":{"id":"deepinfra/ByteDance/Seed-2.0-mini","name":"Seed 2.0 Mini (Deep Infra)","description":"Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":32000},"cost":{"input":0.1,"output":0.4,"cache_read":0.02}},"deepinfra/ByteDance/Seed-2.0-code":{"id":"deepinfra/ByteDance/Seed-2.0-code","name":"Seed 2.0 Code (Deep Infra)","description":"ByteDance Seed coding model for multimodal software engineering and long-running agents","family":"seed","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-02-14","last_updated":"2026-02-14","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":131072},"cost":{"input":0.5,"output":3,"cache_read":0.1}},"deepinfra/stepfun-ai/Step-3.7-Flash":{"id":"deepinfra/stepfun-ai/Step-3.7-Flash","name":"Step 3.7 Flash (Deep Infra)","description":"Newer StepFun flash model for faster agents, coding, and multimodal prompts","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03-01","release_date":"2026-05-29","last_updated":"2026-05-29","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.2,"output":1.15,"cache_read":0.04}},"deepinfra/stepfun-ai/Step-3.5-Flash":{"id":"deepinfra/stepfun-ai/Step-3.5-Flash","name":"Step 3.5 Flash (Deep Infra)","description":"StepFun flash lane for quick multimodal reasoning and coding assistance","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-01-29","last_updated":"2026-02-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.09,"output":0.3,"cache_read":0.02}},"deepinfra/deepseek-ai/DeepSeek-V3":{"id":"deepinfra/deepseek-ai/DeepSeek-V3","name":"DeepSeek-V3 (Deep Infra)","description":"Open DeepSeek MoE chat model for coding, math, and general reasoning","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2024-12-26","last_updated":"2024-12-26","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":8192},"cost":{"input":0.32,"output":0.89}},"deepinfra/deepseek-ai/DeepSeek-V3-0324":{"id":"deepinfra/deepseek-ai/DeepSeek-V3-0324","name":"DeepSeek V3 0324 (Deep Infra)","description":"March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-03-24","last_updated":"2025-03-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":163840},"cost":{"input":0.24,"output":0.9,"cache_read":0.135}},"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Deep Infra)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.06,"output":0.18,"cache_read":0.015}},"deepinfra/deepseek-ai/DeepSeek-V4.1-Flash":{"id":"deepinfra/deepseek-ai/DeepSeek-V4.1-Flash","name":"DeepSeek V4.1 Flash (Deep Infra)","description":"DeepSeek V4.1 Flash model for reasoning and agentic coding","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-09-10","last_updated":"2026-09-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.3,"output":1.2,"cache_read":0.006}},"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"deepinfra/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Deep Infra)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.3,"output":2.6,"cache_read":0.1}},"deepinfra/deepseek-ai/DeepSeek-R1":{"id":"deepinfra/deepseek-ai/DeepSeek-R1","name":"DeepSeek-R1 (Deep Infra)","description":"Classic open reasoning model for transparent math, coding, and deliberate problem solving","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2024-07","release_date":"2025-01-20","last_updated":"2025-05-29","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":163840,"output":32768},"cost":{"input":0.7,"output":2.4}},"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct":{"id":"deepinfra/nvidia/Llama-3.1-Nemotron-70B-Instruct","name":"Llama 3.1 Nemotron 70B Instruct (Deep Infra)","description":"Nemotron model for efficient reasoning, coding, and specialized AI agents","family":"nemotron","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-15","last_updated":"2025-04-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":8192},"cost":{"input":0.6,"output":0.6}},"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B":{"id":"deepinfra/nvidia/Nemotron-3-Nano-30B-A3B","name":"Nemotron 3 Nano 30B A3B (Deep Infra)","description":"Small Nemotron 3 MoE for efficient coding, math, and long-context agents","family":"nemotron","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-12-15","last_updated":"2025-12-15","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.05,"output":0.2,"cache_read":0.025}},"deepinfra/meta-models/Muse-Glimmer-30B":{"id":"deepinfra/meta-models/Muse-Glimmer-30B","name":"Muse Glimmer 30B (Deep Infra)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.3,"output":1.2,"cache_read":0.04}},"deepinfra/zai-org/GLM-4.7-Flash":{"id":"deepinfra/zai-org/GLM-4.7-Flash","name":"GLM-4.7-Flash (Deep Infra)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.06,"output":0.4,"cache_read":0.01}},"deepinfra/thinkingmachines/Inkling-Small":{"id":"deepinfra/thinkingmachines/Inkling-Small","name":"Inkling Small (Deep Infra)","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.45,"output":1.2,"cache_read":0.1}},"deepinfra/thinkingmachines/Inkling":{"id":"deepinfra/thinkingmachines/Inkling","name":"Inkling (Deep Infra)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.95,"output":4.05,"cache_read":0.16}},"deepinfra/meta-llama/Llama-Guard-3-8B":{"id":"deepinfra/meta-llama/Llama-Guard-3-8B","name":"Llama-Guard-3-8B (Deep Infra)","description":"Llama 3.1-based safety classifier for moderating prompts and model responses","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-07-23","last_updated":"2024-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.055,"output":0.055}},"deepinfra/meta-llama/Llama-3.3-70B-Instruct":{"id":"deepinfra/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (Deep Infra)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.1,"output":0.32}},"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct":{"id":"deepinfra/meta-llama/Llama-3.2-11B-Vision-Instruct","name":"Llama-3.2-11B-Vision-Instruct (Deep Infra)","description":"Open multimodal Llama model for image understanding, captioning, and visual QA","family":"llama","attachment":true,"reasoning":false,"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2024-09-25","last_updated":"2024-09-25","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":4096},"cost":{"input":0.345,"output":0.345}},"deepinfra/openai/gpt-oss-20b":{"id":"deepinfra/openai/gpt-oss-20b","name":"GPT OSS 20B (Deep Infra)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.03,"output":0.14}},"deepinfra/openai/gpt-oss-120b":{"id":"deepinfra/openai/gpt-oss-120b","name":"GPT OSS 120B (Deep Infra)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.037,"output":0.17}},"deepinfra/moonshotai/Kimi-K2.5":{"id":"deepinfra/moonshotai/Kimi-K2.5","name":"Kimi K2.5 (Deep Infra)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.45,"output":2.25,"cache_read":0.07}},"deepinfra/tencent/Hy3":{"id":"deepinfra/tencent/Hy3","name":"Hy3 (Deep Infra)","description":"Tencent Hy reasoning model for coding, instruction following, and agent tasks","family":"Hy","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-06","last_updated":"2026-07-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":128000},"cost":{"input":0.14,"output":0.58,"cache_read":0.035}},"moonshot/kimi-k2.7-code-highspeed":{"id":"moonshot/kimi-k2.7-code-highspeed","name":"Kimi K2.7 Code Highspeed","description":"Lower-latency Kimi Code variant for interactive edits and coding-agent loops","family":"kimi-k2","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.9,"output":8,"cache_read":0.38}},"moonshot/kimi-k2.6":{"id":"moonshot/kimi-k2.6","name":"Kimi K2.6","description":"Multimodal Kimi workhorse for agent loops, coding tasks, and visual context","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-04-21","last_updated":"2026-04-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.16}},"moonshot/kimi-k2.7-code":{"id":"moonshot/kimi-k2.7-code","name":"Kimi K2.7 Code","description":"Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-01","release_date":"2026-06-12","last_updated":"2026-06-12","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.95,"output":4,"cache_read":0.19}},"moonshot/kimi-k3":{"id":"moonshot/kimi-k3","name":"Kimi K3","description":"Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work","family":"kimi-k3","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"release_date":"2026-07-16","last_updated":"2026-07-16","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":3,"output":15,"cache_read":0.3}},"azure/gpt-5.1-codex-mini":{"id":"azure/gpt-5.1-codex-mini","name":"GPT-5.1 Codex mini (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"azure/gpt-5.1-codex":{"id":"azure/gpt-5.1-codex","name":"GPT-5.1 Codex (Azure)","description":"Codex GPT for repository edits, code review, and practical software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"azure/gpt-5.2-codex":{"id":"azure/gpt-5.2-codex","name":"GPT-5.2 Codex (Azure)","description":"Code-specialist GPT for repository edits, reviews, and long-running software agents","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"azure/gpt-5.1-codex-max":{"id":"azure/gpt-5.1-codex-max","name":"GPT-5.1 Codex Max (Azure)","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"deepseek/deepseek-v4-flash-vision-exp":{"id":"deepseek/deepseek-v4-flash-vision-exp","name":"DeepSeek V4 Flash Vision Exp","description":"Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work","family":"deepseek-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-21","last_updated":"2026-08-21","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"deepseek/deepseek-v4-flash":{"id":"deepseek/deepseek-v4-flash","name":"DeepSeek V4 Flash","description":"Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":0.44,"output":1.32,"cache_read":0.014}},"deepseek/deepseek-chat":{"id":"deepseek/deepseek-chat","name":"DeepSeek Chat","description":"DeepSeek chat model for instruction following, coding, and analysis","family":"deepseek","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-09","release_date":"2025-12-01","last_updated":"2026-02-28","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":384000},"cost":{"input":0.28,"output":0.42,"cache_read":0.028}},"deepseek/deepseek-v4-pro":{"id":"deepseek/deepseek-v4-pro","name":"DeepSeek V4 Pro","description":"Open MoE flagship with million-token context for coding and long agent runs","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-04-24","last_updated":"2026-04-24","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1000000,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"fireworks_ai/gpt-oss-120b":{"id":"fireworks_ai/gpt-oss-120b","name":"GPT OSS 120B (Fireworks AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6,"cache_read":0.014}},"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813":{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-pro-0813","name":"DeepSeek V4 Pro 0813 (Fireworks AI)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.044}},"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731":{"id":"fireworks_ai/accounts/fireworks/models/deepseek-v4-flash-0731","name":"DeepSeek V4 Flash 0731 (Fireworks AI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.22,"output":0.66,"cache_read":0.007}},"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b":{"id":"fireworks_ai/accounts/fireworks/models/muse-glimmer-30b","name":"Muse Glimmer 30B (Fireworks AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"fireworks_ai/accounts/fireworks/models/inkling":{"id":"fireworks_ai/accounts/fireworks/models/inkling","name":"Inkling (Fireworks AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"amazon/moonshotai.kimi-k2.5":{"id":"amazon/moonshotai.kimi-k2.5","name":"Kimi K2.5 (Amazon Bedrock)","description":"Earlier Kimi frontier model for long-context agents, coding, and multimodal work","family":"kimi-k2","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2026-01","last_updated":"2026-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.6,"output":3}},"amazon/amazon.nova-micro-v1:0@us":{"id":"amazon/amazon.nova-micro-v1:0@us","name":"Nova Micro (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.035,"output":0.14}},"amazon/mistral.pixtral-large-2502-v1:0":{"id":"amazon/mistral.pixtral-large-2502-v1:0","name":"Pixtral Large (25.02) (Amazon Bedrock)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"amazon/amazon.nova-lite-v1:0@us":{"id":"amazon/amazon.nova-lite-v1:0@us","name":"Nova Lite (US)","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.06,"output":0.24}},"amazon/zai.glm-4.7-flash@us":{"id":"amazon/zai.glm-4.7-flash@us","name":"GLM-4.7-Flash (Amazon Bedrock, US)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4}},"amazon/amazon.nova-pro-v1:0@us":{"id":"amazon/amazon.nova-pro-v1:0@us","name":"Nova Pro (US)","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.8,"output":3.2}},"amazon/amazon.nova-lite-v1:0":{"id":"amazon/amazon.nova-lite-v1:0","name":"Nova Lite","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-lite","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.06,"output":0.24}},"amazon/amazon.nova-pro-v1:0":{"id":"amazon/amazon.nova-pro-v1:0","name":"Nova Pro","description":"Flagship model for demanding analysis, coding, and production agent workflows","family":"nova-pro","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":300000,"output":8192},"cost":{"input":0.8,"output":3.2}},"amazon/moonshot.kimi-k2-thinking":{"id":"amazon/moonshot.kimi-k2-thinking","name":"Kimi K2 Thinking (Amazon Bedrock)","description":"Thinking Kimi model for slower research passes, planning, and hard technical questions","family":"kimi-thinking","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2024-08","release_date":"2025-11-06","last_updated":"2025-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":262144},"cost":{"input":0.6,"output":2.5}},"amazon/amazon.nova-micro-v1:0":{"id":"amazon/amazon.nova-micro-v1:0","name":"Nova Micro","description":"Efficient model for low-latency assistance, extraction, and routine automation","family":"nova-micro","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-12-03","last_updated":"2024-12-03","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0.035,"output":0.14}},"amazon/mistral.pixtral-large-2502-v1:0@us":{"id":"amazon/mistral.pixtral-large-2502-v1:0@us","name":"Pixtral Large (25.02) (Amazon Bedrock, US)","description":"Mistral vision-language model for image understanding and multimodal chat","family":"pixtral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2025-04-08","last_updated":"2025-04-08","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":2,"output":6}},"amazon/zai.glm-4.7-flash":{"id":"amazon/zai.glm-4.7-flash","name":"GLM-4.7-Flash (Amazon Bedrock)","description":"Budget GLM lane for fast coding help, routing, and everyday automation","family":"glm-flash","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2026-01-19","last_updated":"2026-01-19","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":200000,"output":131072},"cost":{"input":0.07,"output":0.4}},"openai/gpt-5-nano":{"id":"openai/gpt-5-nano","name":"GPT-5 Nano","description":"Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.05,"output":0.4,"cache_read":0.005}},"openai/gpt-4.1-nano":{"id":"openai/gpt-4.1-nano","name":"GPT-4.1 nano","description":"Tiny GPT-4.1 option for classification, routing, and very high-volume tasks","family":"gpt-nano","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.1,"output":0.4,"cache_read":0.025}},"openai/gpt-5-pro":{"id":"openai/gpt-5-pro","name":"GPT-5 Pro","description":"Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-10-06","last_updated":"2025-10-06","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":272000},"cost":{"input":15,"output":120}},"openai/gpt-5.6-sol":{"id":"openai/gpt-5.6-sol","name":"GPT-5.6 Sol","description":"Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows","family":"gpt-sol","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":4,"output":20,"cache_read":0.4,"cache_write":5,"tiers":[{"input":8,"output":30,"cache_read":0.8,"cache_write":10,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":8,"output":30,"cache_read":0.8,"cache_write":10}}},"openai/gpt-4o-2024-08-06":{"id":"openai/gpt-4o-2024-08-06","name":"GPT-4o (2024-08-06)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-08-06","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-latest":{"id":"openai/gpt-latest","name":"GPT Latest (GPT-6 Astra)","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-6-astra":{"id":"openai/gpt-6-astra","name":"GPT-6 Astra","description":"GPT-6 Astra is OpenAI's most capable model for complex reasoning, coding, computer use, research, and document creation.","family":"gpt-astra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-04-30","release_date":"2026-09-04","last_updated":"2026-09-04","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":10,"output":50,"cache_read":1,"cache_write":12.5,"tiers":[{"input":20,"output":75,"cache_read":2,"cache_write":25,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":20,"output":75,"cache_read":2,"cache_write":25}}},"openai/gpt-5.2-pro":{"id":"openai/gpt-5.2-pro","name":"GPT-5.2 Pro","description":"Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":21,"output":168}},"openai/gpt-4.1-mini":{"id":"openai/gpt-4.1-mini","name":"GPT-4.1 mini","description":"Affordable GPT-4.1 lane for fast coding help and structured extraction","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":0.4,"output":1.6,"cache_read":0.1}},"openai/gpt-5.4":{"id":"openai/gpt-5.4","name":"GPT-5.4","description":"Agent-ready GPT for coding and computer-use workflows at a lower cost","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2.5,"output":15,"cache_read":0.25,"tiers":[{"input":5,"output":22.5,"cache_read":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":5,"output":22.5,"cache_read":0.5}}},"openai/gpt-4-turbo":{"id":"openai/gpt-4-turbo","name":"GPT-4 Turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-12","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":10,"output":30}},"openai/gpt-5.1":{"id":"openai/gpt-5.1","name":"GPT-5.1","description":"Sharper GPT-5 generation for coding, product work, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-09-30","release_date":"2025-11-13","last_updated":"2025-11-13","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o1":{"id":"openai/o1","name":"o1","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2024-12-05","last_updated":"2024-12-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":15,"output":60,"cache_read":7.5}},"openai/gpt-4o":{"id":"openai/gpt-4o","name":"GPT-4o","description":"Omni-era GPT for multimodal chat, practical coding, and general assistants","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-05-13","last_updated":"2024-08-06","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"openai/gpt-5.6-luna":{"id":"openai/gpt-5.6-luna","name":"GPT-5.6 Luna","description":"Cost-efficient GPT-5.6 model for fast, high-volume workloads","family":"gpt-luna","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":0.2,"output":1.2,"cache_read":0.02,"cache_write":0.25,"tiers":[{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":0.4,"output":1.8,"cache_read":0.04,"cache_write":0.5}}},"openai/gpt-5.3-codex":{"id":"openai/gpt-5.3-codex","name":"GPT-5.3 Codex","description":"Coding-optimized GPT model for repository edits, reviews, and agentic software work","family":"gpt-codex","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-02-05","last_updated":"2026-02-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-4o-mini":{"id":"openai/gpt-4o-mini","name":"GPT-4o mini","description":"Small omni GPT for cheap multimodal assistance and production-scale traffic","family":"gpt-mini","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-07-18","last_updated":"2024-07-18","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":0.15,"output":0.6,"cache_read":0.075}},"openai/o1-pro":{"id":"openai/o1-pro","name":"o1-pro","description":"O-series reasoning model for hard analysis, math, coding, and planning","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2023-09","release_date":"2025-03-19","last_updated":"2025-03-19","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":150,"output":600}},"openai/gpt-4.1":{"id":"openai/gpt-4.1","name":"GPT-4.1","description":"Long-lived GPT workhorse for coding, instruction following, and production apps","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-04","release_date":"2025-04-14","last_updated":"2025-04-14","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1047576,"output":32768},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/gpt-5.4-nano":{"id":"openai/gpt-5.4-nano","name":"GPT-5.4 nano","description":"Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation","family":"gpt-nano","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.2,"output":1.25,"cache_read":0.02}},"openai/gpt-5.5-pro":{"id":"openai/gpt-5.5-pro","name":"GPT-5.5 Pro","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":6,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":6}}},"openai/gpt-5.4-mini":{"id":"openai/gpt-5.4-mini","name":"GPT-5.4 mini","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/gpt-3.5-turbo":{"id":"openai/gpt-3.5-turbo","name":"GPT-3.5-turbo","description":"Compact GPT model for low-latency assistance and high-volume workloads","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2021-09-01","release_date":"2023-03-01","last_updated":"2023-11-06","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":16385,"output":4096},"cost":{"input":0.5,"output":1.5}},"openai/gpt-5-mini":{"id":"openai/gpt-5-mini","name":"GPT-5 Mini","description":"Small GPT-5 for responsive agents, coding help, and everyday automation","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.25,"output":2,"cache_read":0.025}},"openai/gpt-5.4-pro":{"id":"openai/gpt-5.4-pro","name":"GPT-5.4 Pro","description":"More exact GPT-5.4 tier for demanding professional reasoning and agent tasks","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-08-31","release_date":"2026-03-05","last_updated":"2026-03-05","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":6,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":6}}},"openai/gpt-5.6-terra":{"id":"openai/gpt-5.6-terra","name":"GPT-5.6 Terra","description":"Balanced GPT-5.6 model for capable, cost-efficient everyday work","family":"gpt-terra","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh","max"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2026-02-16","release_date":"2026-07-09","last_updated":"2026-07-09","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":2,"output":12,"cache_read":0.2,"cache_write":2.5,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":5,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":5}}},"openai/gpt-4":{"id":"openai/gpt-4","name":"GPT-4","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-11","release_date":"2023-11-06","last_updated":"2024-04-09","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":8191,"output":8192},"cost":{"input":30,"output":60}},"openai/gpt-5.2":{"id":"openai/gpt-5.2","name":"GPT-5.2","description":"Reliable GPT generation for broad coding, writing, and tool-assisted product work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2025-12-11","last_updated":"2025-12-11","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.75,"output":14,"cache_read":0.175}},"openai/gpt-5":{"id":"openai/gpt-5","name":"GPT-5","description":"Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-09-30","release_date":"2025-08-07","last_updated":"2025-08-07","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":1.25,"output":10,"cache_read":0.125}},"openai/o4-mini":{"id":"openai/o4-mini","name":"o4-mini","description":"Fast o-series model for compact reasoning, coding, and tool use","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["image","text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.275}},"openai/gpt-mini-latest":{"id":"openai/gpt-mini-latest","name":"GPT Mini Latest (GPT-5.4 mini)","description":"Strong small GPT for coding subagents, quick tool use, and high-volume work","family":"gpt-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-08-31","release_date":"2026-03-17","last_updated":"2026-03-17","modalities":{"input":["pdf","image","text"],"output":["text"]},"open_weights":false,"limit":{"context":400000,"input":272000,"output":128000},"cost":{"input":0.75,"output":4.5,"cache_read":0.075}},"openai/o3-mini":{"id":"openai/o3-mini","name":"o3-mini","description":"Smaller o-series reasoner for economical coding, math, and planning tasks","family":"o-mini","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2024-12-20","last_updated":"2025-01-29","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":1.1,"output":4.4,"cache_read":0.55}},"openai/gpt-pro-latest":{"id":"openai/gpt-pro-latest","name":"GPT Pro Latest (GPT-5.5 Pro)","description":"Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding","family":"gpt-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":30,"output":180,"cache_read":3,"tiers":[{"input":60,"output":270,"cache_read":6,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":60,"output":270,"cache_read":6}}},"openai/o3":{"id":"openai/o3","name":"o3","description":"Deliberate o-series reasoner for hard math, coding, and multi-step analysis","family":"o","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-04-16","last_updated":"2025-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":2,"output":8,"cache_read":0.5}},"openai/o3-pro":{"id":"openai/o3-pro","name":"o3-pro","description":"High-effort o3 tier for difficult technical reasoning and careful answers","family":"o-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2024-05","release_date":"2025-06-10","last_updated":"2025-06-10","modalities":{"input":["text","pdf","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":100000},"cost":{"input":20,"output":80}},"openai/gpt-5.5":{"id":"openai/gpt-5.5","name":"GPT-5.5","description":"Default frontier GPT for coding, computer use, research, and knowledge work","family":"gpt","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":false,"knowledge":"2025-12-01","release_date":"2026-04-23","last_updated":"2026-04-23","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1050000,"input":922000,"output":128000},"cost":{"input":5,"output":30,"cache_read":0.5,"tiers":[{"input":10,"output":45,"cache_read":1,"tier":{"type":"context","size":272000}}],"context_over_200k":{"input":10,"output":45,"cache_read":1}}},"openai/gpt-4o-2024-11-20":{"id":"openai/gpt-4o-2024-11-20","name":"GPT-4o (2024-11-20)","description":"GPT model for general reasoning, writing, coding, and tool-assisted tasks","family":"gpt","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2023-09","release_date":"2024-11-20","last_updated":"2024-11-20","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":16384},"cost":{"input":2.5,"output":10,"cache_read":1.25}},"cohere/command-r-plus-08-2024":{"id":"cohere/command-r-plus-08-2024","name":"Command R+","description":"Cohere's RAG workhorse for long-context enterprise search and tool use","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":2.5,"output":10}},"cohere/command-a-03-2025":{"id":"cohere/command-a-03-2025","name":"Command A","description":"Cohere command model for multilingual enterprise agents, tools, and chat","family":"command-a","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2025-03-13","last_updated":"2025-03-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":288000,"output":8000},"cost":{"input":2.5,"output":10}},"cohere/command-r7b-12-2024":{"id":"cohere/command-r7b-12-2024","name":"Command R7B","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-12-02","last_updated":"2024-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":132000,"output":4000},"cost":{"input":0.0375,"output":0.15}},"cohere/command-r-08-2024":{"id":"cohere/command-r-08-2024","name":"Command R","description":"Cohere retrieval model for long-context chat and enterprise RAG workflows","family":"command-r","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-06-01","release_date":"2024-08-30","last_updated":"2024-08-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4000},"cost":{"input":0.15,"output":0.6}},"xai/grok-4.3":{"id":"xai/grok-4.3","name":"Grok 4.3","description":"xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-17","last_updated":"2026-04-17","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-4.20-0309-reasoning":{"id":"xai/grok-4.20-0309-reasoning","name":"Grok 4.20 (Reasoning)","description":"Reasoning Grok for document-heavy analysis and long-horizon tool use","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"xai/grok-latest":{"id":"xai/grok-latest","name":"Grok Latest (Grok 4.6)","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai/grok-4.5":{"id":"xai/grok-4.5","name":"Grok 4.5","description":"xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-08","last_updated":"2026-07-08","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.3,"tiers":[{"input":4,"output":12,"cache_read":0.6,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":0.6}}},"xai/grok-build-0.1":{"id":"xai/grok-build-0.1","name":"Grok Build 0.1","description":"Fast Grok coding model tuned for agentic engineering and iterative edits","family":"grok-build","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-16","last_updated":"2026-04-16","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":256000,"output":256000},"cost":{"input":1,"output":2,"cache_read":0.2,"tiers":[{"input":2,"output":4,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2,"output":4,"cache_read":0.4}}},"xai/grok-4.6":{"id":"xai/grok-4.6","name":"Grok 4.6","description":"xAI's frontier model for long-running agents, coding, knowledge work, and visual projects","family":"grok","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-02-01","release_date":"2026-08-12","last_updated":"2026-08-12","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":500000,"output":500000},"cost":{"input":2,"output":6,"cache_read":0.5,"tiers":[{"input":4,"output":12,"cache_read":1,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":12,"cache_read":1}}},"xai/grok-4.20-0309-non-reasoning":{"id":"xai/grok-4.20-0309-non-reasoning","name":"Grok 4.20 (Non-Reasoning)","description":"Grok model for agentic tool use, reasoning, coding, and live assistance","family":"grok","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-09","last_updated":"2026-03-09","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":1000000,"output":30000},"cost":{"input":1.25,"output":2.5,"cache_read":0.2,"tiers":[{"input":2.5,"output":5,"cache_read":0.4,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":2.5,"output":5,"cache_read":0.4}}},"zai/glm-4.7":{"id":"zai/glm-4.7","name":"GLM-4.7","description":"Mature GLM model for dependable coding, reasoning, and structured agent tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-22","last_updated":"2025-12-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6":{"id":"zai/glm-4.6","name":"GLM-4.6","description":"Late GLM-4 workhorse for coding agents, reasoning, and structured tasks","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-09-30","last_updated":"2025-09-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":0.6,"output":2.2,"cache_read":0.11}},"zai/glm-4.6v":{"id":"zai/glm-4.6v","name":"GLM-4.6V","description":"GLM vision model for visual reasoning, documents, and multimodal agents","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"knowledge":"2025-04","release_date":"2025-12-08","last_updated":"2025-12-08","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.3,"output":0.9,"cache_read":0.05}},"zai/glm-5.2":{"id":"zai/glm-5.2","name":"GLM-5.2","description":"Open flagship GLM for long-horizon coding agents and million-token context work","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-06-13","last_updated":"2026-06-13","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5.3-flash":{"id":"zai/glm-5.3-flash","name":"GLM-5.3-Flash","description":"Native multimodal GLM model for efficient coding and long-horizon agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-26","last_updated":"2026-08-26","modalities":{"input":["text","image","video"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":0.15,"output":0.5,"cache_read":0.03}},"zai/glm-5":{"id":"zai/glm-5","name":"GLM-5","description":"General GLM flagship for coding, analysis, and tool-heavy engineering workflows","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-02-12","last_updated":"2026-02-12","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1,"output":3.2,"cache_read":0.2}},"zai/glm-5.1":{"id":"zai/glm-5.1","name":"GLM-5.1","description":"Strong GLM coding model for agentic engineering, terminals, and repository generation","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-07","last_updated":"2026-04-07","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":202752,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5-turbo":{"id":"zai/glm-5-turbo","name":"GLM-5-Turbo","description":"Faster GLM-5 lane for coding agents that need lower latency","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"zai/glm-5.3":{"id":"zai/glm-5.3","name":"GLM-5.3","description":"Flagship GLM model for long-horizon coding, agents, and complex project delivery","family":"glm","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-14","last_updated":"2026-08-14","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":131072},"cost":{"input":1.4,"output":4.4,"cache_read":0.26}},"zai/glm-5v-turbo":{"id":"zai/glm-5v-turbo","name":"GLM-5V-Turbo","description":"Fast GLM vision model for screenshots, documents, and multimodal agent tasks","family":"glm","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-01","last_updated":"2026-04-01","modalities":{"input":["image","text","video"],"output":["text"]},"open_weights":false,"limit":{"context":202752,"output":131072},"cost":{"input":1.2,"output":4,"cache_read":0.24}},"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731":{"id":"together_ai/deepseek-ai/DeepSeek-V4-Flash-0731","name":"DeepSeek V4 Flash 0731 (Together AI)","description":"Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding","family":"deepseek-flash","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2026-07-31","last_updated":"2026-07-31","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":0.14,"output":0.28,"cache_read":0.03}},"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813":{"id":"together_ai/deepseek-ai/DeepSeek-V4-Pro-0813","name":"DeepSeek V4 Pro 0813 (Together AI)","description":"DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes","family":"deepseek-thinking","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","low","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-08-12","last_updated":"2026-08-22","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":1048576,"output":384000},"cost":{"input":1.32,"output":3.96,"cache_read":0.13}},"together_ai/meta-models/Muse-Glimmer-30B":{"id":"together_ai/meta-models/Muse-Glimmer-30B","name":"Muse Glimmer 30B (Together AI)","description":"Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.","family":"muse","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high","xhigh"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-01-04","release_date":"2026-08-10","last_updated":"2026-08-10","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":131072},"cost":{"input":0.35,"output":1.5,"cache_read":0.04}},"together_ai/thinkingmachines/Inkling-Small":{"id":"together_ai/thinkingmachines/Inkling-Small","name":"Inkling Small (Together AI)","description":"Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":false,"temperature":true,"release_date":"2026-07-30","last_updated":"2026-07-30","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":0.5,"output":1.2,"cache_read":0.1}},"together_ai/thinkingmachines/Inkling":{"id":"together_ai/thinkingmachines/Inkling","name":"Inkling (Together AI)","description":"Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio","family":"ling","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","minimal","low","medium","high","max"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-07-15","last_updated":"2026-07-15","modalities":{"input":["text","image","audio"],"output":["text"]},"open_weights":true,"limit":{"context":524288,"output":1048576},"cost":{"input":1,"output":4.05,"cache_read":0.17}},"together_ai/openai/gpt-oss-20b":{"id":"together_ai/openai/gpt-oss-20b","name":"GPT OSS 20B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.05,"output":0.2}},"together_ai/openai/gpt-oss-120b":{"id":"together_ai/openai/gpt-oss-120b","name":"GPT OSS 120B (Together AI)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.15,"output":0.6}},"mistral/devstral-2512":{"id":"mistral/devstral-2512","name":"Devstral 2","description":"Mistral's coding-agent model for repository work, terminal tasks, and software fixes","family":"devstral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-09","last_updated":"2025-12-09","modalities":{"input":["text","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2,"cache_read":0.04}},"mistral/magistral-medium-latest":{"id":"mistral/magistral-medium-latest","name":"Magistral Medium (latest)","description":"Mistral reasoning model for transparent analysis, math, and complex decisions","family":"magistral-medium","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2025-03-17","last_updated":"2025-03-20","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":262144,"output":16384},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-medium-latest":{"id":"mistral/mistral-medium-latest","name":"Mistral Medium (latest)","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-medium-2604":{"id":"mistral/mistral-medium-2604","name":"Mistral Medium 3.5","description":"Balanced Mistral model for enterprise assistants, multilingual work, and tools","family":"mistral-medium","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-04-29","last_updated":"2026-04-29","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":1.5,"output":7.5,"cache_read":0.15}},"mistral/mistral-large-2512":{"id":"mistral/mistral-large-2512","name":"Mistral Large 3","description":"Mistral's largest general model for enterprise agents, coding, and multilingual reasoning","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.5,"output":1.5,"cache_read":0.05}},"mistral/devstral-medium-latest":{"id":"mistral/devstral-medium-latest","name":"Devstral 2 (latest)","description":"Mistral coding agent model for repository tasks and software engineering workflows","family":"devstral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-12","release_date":"2025-12-02","last_updated":"2025-12-02","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":0.4,"output":2}},"mistral/voxtral-small-latest":{"id":"mistral/voxtral-small-latest","name":"Voxtral Small (latest)","description":"Instruct model with native audio input for speech understanding and tool use","family":"voxtral","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-07-15","last_updated":"2025-07-15","modalities":{"input":["text","audio"],"output":["text"]},"open_weights":true,"limit":{"context":32768,"output":32000},"cost":{"input":0.1,"output":0.4}},"mistral/mistral-small-latest":{"id":"mistral/mistral-small-latest","name":"Mistral Small (latest)","description":"Efficient Mistral model for fast chat, extraction, and production assistants","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/mistral-large-latest":{"id":"mistral/mistral-large-latest","name":"Mistral Large (latest)","description":"Flagship Mistral model for advanced reasoning, coding, and multilingual work","family":"mistral-large","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-11","release_date":"2024-11-01","last_updated":"2025-12-02","modalities":{"input":["text","image","pdf"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":262144},"cost":{"input":2,"output":6,"cache_read":0.2}},"mistral/mistral-small-2603":{"id":"mistral/mistral-small-2603","name":"Mistral Small 4","description":"Fast Mistral production model for chat, extraction, and cost-sensitive agents","family":"mistral-small","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["none","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-06","release_date":"2026-03-16","last_updated":"2026-03-16","modalities":{"input":["text","image"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":256000},"cost":{"input":0.15,"output":0.6,"cache_read":0.015}},"mistral/mistral-medium-2505":{"id":"mistral/mistral-medium-2505","name":"Mistral Medium 3","description":"Mistral model for multilingual chat, reasoning, and tool-assisted workflows","family":"mistral-medium","attachment":true,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-05","release_date":"2025-05-07","last_updated":"2025-05-07","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":131072,"output":131072},"cost":{"input":0.4,"output":2}},"mistral/codestral-latest":{"id":"mistral/codestral-latest","name":"Codestral (latest)","description":"Mistral code model for completions, refactors, and developer IDE workflows","family":"codestral","attachment":false,"reasoning":false,"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2024-10","release_date":"2024-05-29","last_updated":"2025-01-04","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":256000,"output":4096},"cost":{"input":0.3,"output":0.9,"cache_read":0.03}},"vertex/gemini-pro-latest":{"id":"vertex/gemini-pro-latest","name":"Gemini Pro Latest (Gemini 3.1 Pro Preview, Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25}}},"vertex/gemini-3.1-flash-lite-image":{"id":"vertex/gemini-3.1-flash-lite-image","name":"Nano Banana 2 Lite (Vertex AI)","description":"Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-06-30","last_updated":"2026-06-30","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":4096},"cost":{"input":0.25,"output":1.5}},"vertex/gemini-3.7-flash@eu":{"id":"vertex/gemini-3.7-flash@eu","name":"Gemini 3.7 Flash (Vertex AI, EU)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-2.5-flash-image":{"id":"vertex/gemini-2.5-flash-image","name":"Nano Banana (Vertex AI)","description":"Nano Banana image model for fast generation, edits, and character-consistent assets","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2024-06","release_date":"2025-08-26","last_updated":"2025-08-26","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":32768,"output":32768},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":1}},"vertex/gemini-3-pro-image":{"id":"vertex/gemini-3-pro-image","name":"Nano Banana Pro (Vertex AI)","description":"Nano Banana Pro for higher-fidelity image generation and design-heavy edits","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["text","image"],"output":["text","image"]},"open_weights":false,"limit":{"context":65536,"output":32768},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2}},"vertex/gemini-3.1-pro-preview":{"id":"vertex/gemini-3.1-pro-preview","name":"Gemini 3.1 Pro Preview (Vertex AI)","description":"Reasoning-first Gemini preview for agentic coding and complex problem solving","family":"gemini-pro","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-02-19","last_updated":"2026-02-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":2,"output":12,"reasoning":12,"cache_read":0.2,"cache_write":0.375,"input_audio":2,"tiers":[{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25,"tier":{"type":"context","size":200000}}],"context_over_200k":{"input":4,"output":18,"cache_read":0.4,"cache_write":0.25}}},"vertex/gemini-3.8-flash@eu":{"id":"vertex/gemini-3.8-flash@eu","name":"Gemini 3.8 Flash (Vertex AI, EU)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.6-flash":{"id":"vertex/gemini-3.6-flash","name":"Gemini 3.6 Flash (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.1-flash-lite":{"id":"vertex/gemini-3.1-flash-lite","name":"Gemini 3.1 Flash Lite (Vertex AI)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.1-flash-lite@us":{"id":"vertex/gemini-3.1-flash-lite@us","name":"Gemini 3.1 Flash Lite (Vertex AI, US)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.6-flash@eu":{"id":"vertex/gemini-3.6-flash@eu","name":"Gemini 3.6 Flash (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.8-flash@us":{"id":"vertex/gemini-3.8-flash@us","name":"Gemini 3.8 Flash (Vertex AI, US)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.6-flash@us":{"id":"vertex/gemini-3.6-flash@us","name":"Gemini 3.6 Flash (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.7-flash@us":{"id":"vertex/gemini-3.7-flash@us","name":"Gemini 3.7 Flash (Vertex AI, US)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash":{"id":"vertex/gemini-3.5-flash","name":"Gemini 3.5 Flash (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3.1-flash-image":{"id":"vertex/gemini-3.1-flash-image","name":"Nano Banana 2 (Vertex AI)","description":"Image model for prompt-driven generation, editing, and visual design workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","high"]}],"tool_call":false,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-28","last_updated":"2026-05-28","modalities":{"input":["image","text"],"output":["text","image"]},"open_weights":false,"limit":{"context":131072,"output":32768},"cost":{"input":0.5,"output":3}},"vertex/gemini-3.5-flash-lite":{"id":"vertex/gemini-3.5-flash-lite","name":"Gemini 3.5 Flash Lite (Vertex AI)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.5-flash-lite@us":{"id":"vertex/gemini-3.5-flash-lite@us","name":"Gemini 3.5 Flash Lite (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.5-flash-lite@eu":{"id":"vertex/gemini-3.5-flash-lite@eu","name":"Gemini 3.5 Flash Lite (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-07-21","last_updated":"2026-07-21","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.3,"output":2.5,"reasoning":2.5,"cache_read":0.03,"cache_write":0.083333,"input_audio":0.3}},"vertex/gemini-3.5-flash@us":{"id":"vertex/gemini-3.5-flash@us","name":"Gemini 3.5 Flash (Vertex AI, US)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"vertex/gemini-3-flash-preview":{"id":"vertex/gemini-3-flash-preview","name":"Gemini 3 Flash Preview (Vertex AI)","description":"New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2025-12-17","last_updated":"2025-12-17","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.5,"output":3,"reasoning":3,"cache_read":0.05,"cache_write":0.083333,"input_audio":1}},"vertex/gemini-3.8-flash":{"id":"vertex/gemini-3.8-flash","name":"Gemini 3.8 Flash (Vertex AI)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.1-flash-lite@eu":{"id":"vertex/gemini-3.1-flash-lite@eu","name":"Gemini 3.1 Flash Lite (Vertex AI, EU)","description":"Low-latency Gemini model for high-volume multimodal and agent workloads","family":"gemini-flash-lite","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-07","last_updated":"2026-05-07","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.25,"output":1.5,"reasoning":1.5,"cache_read":0.025,"cache_write":0.083333,"input_audio":0.5}},"vertex/gemini-3.7-flash":{"id":"vertex/gemini-3.7-flash","name":"Gemini 3.7 Flash (Vertex AI)","description":"High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2026-03","release_date":"2026-08-13","last_updated":"2026-08-13","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-flash-latest":{"id":"vertex/gemini-flash-latest","name":"Gemini Flash Latest (Gemini 3.8 Flash, Vertex AI)","description":"Google's most intelligent Flash model, engineered for long-horizon software engineering, autonomous agents, and complex enterprise workflows","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2026-09-02","last_updated":"2026-09-02","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":0.75,"output":3.75,"reasoning":3.75,"cache_read":0.075,"cache_write":0.041667,"input_audio":0.75}},"vertex/gemini-3.5-flash@eu":{"id":"vertex/gemini-3.5-flash@eu","name":"Gemini 3.5 Flash (Vertex AI, EU)","description":"Fast Gemini model balancing multimodal reasoning, tool use, and cost","family":"gemini-flash","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"knowledge":"2025-01","release_date":"2026-05-19","last_updated":"2026-05-19","modalities":{"input":["text","image","video","audio","pdf"],"output":["text"]},"open_weights":false,"limit":{"context":1048576,"output":65536},"cost":{"input":1.5,"output":9,"reasoning":9,"cache_read":0.15,"cache_write":0.083333,"input_audio":3}},"cerebras/gpt-oss-120b":{"id":"cerebras/gpt-oss-120b","name":"GPT OSS 120B (Cerebras)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"structured_output":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.35,"output":0.75,"cache_read":0.35}},"ionos/meta-llama/Llama-3.3-70B-Instruct":{"id":"ionos/meta-llama/Llama-3.3-70B-Instruct","name":"Llama-3.3-70B-Instruct (IONOS)","description":"Popular open Llama workhorse for multilingual chat, coding, and self-hosting","family":"llama","attachment":false,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2023-12","release_date":"2024-12-06","last_updated":"2024-12-06","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":128000,"output":4096},"cost":{"input":0.75504,"output":0.75504}},"ionos/openai/gpt-oss-120b":{"id":"ionos/openai/gpt-oss-120b","name":"GPT OSS 120B (IONOS)","description":"Open GPT reasoning model for self-hosted agents and controllable deployments","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0.17424,"output":0.75504}},"perplexityai/sonar":{"id":"perplexityai/sonar","name":"Sonar","description":"Fast web-grounded Sonar for current answers, citations, and lightweight retrieval","family":"sonar","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":127072,"output":4096},"cost":{"input":1,"output":1}},"perplexityai/sonar-reasoning-pro":{"id":"perplexityai/sonar-reasoning-pro","name":"Sonar Reasoning Pro","description":"Web-grounded Sonar for multi-step research questions that need cited reasoning","family":"sonar-reasoning","attachment":true,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":4096},"cost":{"input":2,"output":8}},"perplexityai/sonar-pro":{"id":"perplexityai/sonar-pro","name":"Sonar Pro","description":"Deeper Sonar search model with broader retrieval and stronger synthesis","family":"sonar-pro","attachment":true,"reasoning":false,"tool_call":false,"structured_output":false,"temperature":true,"knowledge":"2025-09-01","release_date":"2024-01-01","last_updated":"2025-09-01","modalities":{"input":["text","image"],"output":["text"]},"open_weights":false,"limit":{"context":200000,"output":8192},"cost":{"input":3,"output":15}},"perplexityai/sonar-deep-research":{"id":"perplexityai/sonar-deep-research","name":"Sonar Deep Research","description":"Sonar search model for autonomous research and citation-backed long-form reports","family":"sonar","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["minimal","low","medium","high"]}],"tool_call":false,"structured_output":false,"temperature":false,"knowledge":"2025-01","release_date":"2025-02-01","last_updated":"2025-09-01","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":32768},"cost":{"input":2,"output":8,"reasoning":3}}}},"lmstudio":{"id":"lmstudio","env":["LMSTUDIO_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:1234/v1","name":"LMStudio","doc":"https://lmstudio.ai/models","models":{"qwen/qwen3-coder-30b":{"id":"qwen/qwen3-coder-30b","name":"Qwen3 Coder 30B","description":"Qwen coding model for software agents, repository edits, and code reasoning","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-23","last_updated":"2025-07-23","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":65536},"cost":{"input":0,"output":0}},"qwen/qwen3-30b-a3b-2507":{"id":"qwen/qwen3-30b-a3b-2507","name":"Qwen3 30B A3B 2507","description":"Qwen instruction model for multilingual chat, reasoning, and tool use","family":"qwen","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"knowledge":"2025-04","release_date":"2025-07-30","last_updated":"2025-07-30","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":262144,"output":16384},"cost":{"input":0,"output":0}},"openai/gpt-oss-20b":{"id":"openai/gpt-oss-20b","name":"GPT OSS 20B","description":"Open-weight GPT model for self-hosted reasoning and instruction-following workloads","family":"gpt-oss","attachment":false,"reasoning":true,"reasoning_options":[{"type":"effort","values":["low","medium","high"]}],"tool_call":true,"temperature":true,"release_date":"2025-08-05","last_updated":"2025-08-05","modalities":{"input":["text"],"output":["text"]},"open_weights":true,"limit":{"context":131072,"output":32768},"cost":{"input":0,"output":0}}}},"lynkr":{"id":"lynkr","env":["LYNKR_API_KEY"],"npm":"@ai-sdk/openai-compatible","api":"http://127.0.0.1:8081/v1","name":"Lynkr","doc":"https://github.com/Fast-Editor/Lynkr","models":{"lynkr-auto":{"id":"lynkr-auto","name":"Lynkr Auto (complexity routing)","description":"Virtual model: Lynkr scores each request on complexity and routes it to the tier model the user configured (local Ollama/llama.cpp for simple requests, configured cloud providers for complex ones).","family":"auto","attachment":false,"reasoning":false,"tool_call":true,"temperature":true,"release_date":"2025-12-03","last_updated":"2026-07-11","modalities":{"input":["text"],"output":["text"]},"open_weights":false,"limit":{"context":128000,"output":8192},"cost":{"input":0,"output":0}}}}}